summaryrefslogtreecommitdiff
path: root/fs
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-09-21 10:30:20 +0200
committerMark Brown <broonie@kernel.org>2026-09-21 10:30:20 +0200
commitce61374442ff7e350a2c8006357aa6a2c550c08d (patch)
tree5dd3c60e8d6bc9bd06e8d391465a1be52f277c1f /fs
parent2a945b9b0925f8766ce52986edfb1fb0b8594134 (diff)
parent55d05b596923df9445a9f6179b8807d6d4c5a50d (diff)
downloadlinux-next-ce61374442ff7e350a2c8006357aa6a2c550c08d.tar.gz
linux-next-ce61374442ff7e350a2c8006357aa6a2c550c08d.zip
Merge branch 'for-next' of ssh://git@gitolite.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git
Diffstat (limited to 'fs')
-rw-r--r--fs/btrfs/Kconfig1
-rw-r--r--fs/btrfs/bio.c163
-rw-r--r--fs/btrfs/bio.h9
-rw-r--r--fs/btrfs/block-group.c498
-rw-r--r--fs/btrfs/block-group.h16
-rw-r--r--fs/btrfs/btrfs_inode.h22
-rw-r--r--fs/btrfs/compression.c67
-rw-r--r--fs/btrfs/ctree.c4
-rw-r--r--fs/btrfs/delalloc-space.c13
-rw-r--r--fs/btrfs/delayed-inode.c115
-rw-r--r--fs/btrfs/delayed-inode.h22
-rw-r--r--fs/btrfs/dev-replace.c10
-rw-r--r--fs/btrfs/dir-item.c41
-rw-r--r--fs/btrfs/dir-item.h5
-rw-r--r--fs/btrfs/disk-io.c126
-rw-r--r--fs/btrfs/extent-io-tree.c2
-rw-r--r--fs/btrfs/extent-tree.c10
-rw-r--r--fs/btrfs/extent_io.c247
-rw-r--r--fs/btrfs/extent_map.c8
-rw-r--r--fs/btrfs/file-item.c46
-rw-r--r--fs/btrfs/free-space-cache.c1344
-rw-r--r--fs/btrfs/free-space-cache.h29
-rw-r--r--fs/btrfs/fs.c2
-rw-r--r--fs/btrfs/fs.h4
-rw-r--r--fs/btrfs/inode.c342
-rw-r--r--fs/btrfs/ioctl.c41
-rw-r--r--fs/btrfs/ordered-data.c27
-rw-r--r--fs/btrfs/qgroup.c126
-rw-r--r--fs/btrfs/qgroup.h16
-rw-r--r--fs/btrfs/raid56.c57
-rw-r--r--fs/btrfs/relocation.c2
-rw-r--r--fs/btrfs/send.c2
-rw-r--r--fs/btrfs/space-info.c2
-rw-r--r--fs/btrfs/space-info.h4
-rw-r--r--fs/btrfs/super.c48
-rw-r--r--fs/btrfs/sysfs.c129
-rw-r--r--fs/btrfs/tests/extent-io-tests.c129
-rw-r--r--fs/btrfs/transaction.c64
-rw-r--r--fs/btrfs/transaction.h31
-rw-r--r--fs/btrfs/tree-checker.c83
-rw-r--r--fs/btrfs/tree-log.c11
-rw-r--r--fs/btrfs/verity.c31
-rw-r--r--fs/btrfs/volumes.c55
-rw-r--r--fs/btrfs/volumes.h1
-rw-r--r--fs/btrfs/zlib.c10
-rw-r--r--fs/btrfs/zoned.c21
-rw-r--r--fs/btrfs/zstd.c73
47 files changed, 1079 insertions, 3030 deletions
diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig
index 4b10d78ed99b..0b8d8905e38e 100644
--- a/fs/btrfs/Kconfig
+++ b/fs/btrfs/Kconfig
@@ -1,4 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
+# misc-next marker
config BTRFS_FS
tristate "Btrfs filesystem support"
diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c
index cc0bd03048ba..2c6234f6182f 100644
--- a/fs/btrfs/bio.c
+++ b/fs/btrfs/bio.c
@@ -27,15 +27,9 @@ struct btrfs_failed_bio {
atomic_t repair_count;
};
-/* Is this a data path I/O that needs storage layer checksum and repair? */
-static inline bool is_data_bbio(const struct btrfs_bio *bbio)
-{
- return bbio->inode && is_data_inode(bbio->inode);
-}
-
static bool bbio_has_ordered_extent(const struct btrfs_bio *bbio)
{
- return is_data_bbio(bbio) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE;
+ return is_data_inode(bbio->inode) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE;
}
/*
@@ -103,7 +97,6 @@ static struct btrfs_bio *btrfs_split_bio(struct btrfs_fs_info *fs_info,
bbio->can_use_append = orig_bbio->can_use_append;
bbio->is_scrub = orig_bbio->is_scrub;
bbio->is_remap = orig_bbio->is_remap;
- bbio->async_csum = orig_bbio->async_csum;
atomic_inc(&orig_bbio->pending_ios);
return bbio;
@@ -114,9 +107,6 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status)
/* Make sure we're already in task context. */
ASSERT(in_task());
- if (bbio->async_csum)
- wait_for_completion(&bbio->csum_done);
-
bbio->bio.bi_status = status;
if (bbio->bio.bi_pool == &btrfs_clone_bioset) {
struct btrfs_bio *orig_bbio = bbio->private;
@@ -180,30 +170,13 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio,
struct btrfs_failed_bio *fbio = repair_bbio->private;
struct btrfs_inode *inode = repair_bbio->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
- /*
- * We can not move forward the saved_iter, as it will be later
- * utilized by repair_bbio again.
- */
- struct bvec_iter saved_iter = repair_bbio->saved_iter;
- const u32 step = min(fs_info->sectorsize, PAGE_SIZE);
- const u64 logical = repair_bbio->saved_iter.bi_sector << SECTOR_SHIFT;
- const u32 nr_steps = repair_bbio->saved_iter.bi_size / step;
int mirror = repair_bbio->mirror_num;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
- phys_addr_t paddr;
- unsigned int slot = 0;
- /* Repair bbio should be eaxctly one block sized. */
+ /* Repair bbio should be exactly one block sized. */
ASSERT(repair_bbio->saved_iter.bi_size == fs_info->sectorsize);
- btrfs_bio_for_each_block(paddr, &repair_bbio->bio, &saved_iter, step) {
- ASSERT(slot < nr_steps);
- paddrs[slot] = paddr;
- slot++;
- }
-
if (repair_bbio->bio.bi_status ||
- !btrfs_data_csum_ok(repair_bbio, dev, 0, paddrs)) {
+ !btrfs_bio_data_csum_ok(repair_bbio, &repair_bbio->saved_iter, dev)) {
bio_reset(&repair_bbio->bio, NULL, REQ_OP_READ);
repair_bbio->bio.bi_iter = repair_bbio->saved_iter;
@@ -220,9 +193,8 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio,
do {
mirror = prev_repair_mirror(fbio, mirror);
- btrfs_repair_io_failure(fs_info, btrfs_ino(inode),
- repair_bbio->file_offset, fs_info->sectorsize,
- logical, paddrs, step, mirror);
+ btrfs_repair_bbio_failure(repair_bbio, &repair_bbio->saved_iter,
+ fs_info->sectorsize, mirror);
} while (mirror != fbio->bbio->mirror_num);
done:
@@ -238,25 +210,21 @@ done:
* read succeeded to restore the redundancy.
*/
static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio,
- u32 bio_offset,
- phys_addr_t paddrs[],
+ const struct bvec_iter *orig_iter,
struct btrfs_failed_bio *fbio)
{
struct btrfs_inode *inode = failed_bbio->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
- const u32 sectorsize = fs_info->sectorsize;
- const u32 step = min(fs_info->sectorsize, PAGE_SIZE);
- const u32 nr_steps = sectorsize / step;
- /*
- * For bs > ps cases, the saved_iter can be partially moved forward.
- * In that case we should round it down to the block boundary.
- */
- const u64 logical = round_down(failed_bbio->saved_iter.bi_sector << SECTOR_SHIFT,
- sectorsize);
struct btrfs_bio *repair_bbio;
struct bio *repair_bio;
+ struct bvec_iter iter = *orig_iter;
+ const u32 sectorsize = fs_info->sectorsize;
+ const u32 bio_offset = ((iter.bi_sector - failed_bbio->saved_iter.bi_sector) <<
+ SECTOR_SHIFT);
+ const u64 logical = (iter.bi_sector << SECTOR_SHIFT);
int num_copies;
int mirror;
+ u32 cur = 0;
btrfs_debug(fs_info, "repair read error: read error at %llu",
failed_bbio->file_offset + bio_offset);
@@ -277,17 +245,21 @@ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio,
atomic_inc(&fbio->repair_count);
- repair_bio = bio_alloc_bioset(NULL, nr_steps, REQ_OP_READ, GFP_NOFS,
- &btrfs_repair_bioset);
+ repair_bio = bio_alloc_bioset(NULL, max(1, sectorsize >> PAGE_SHIFT),
+ REQ_OP_READ, GFP_NOFS, &btrfs_repair_bioset);
repair_bio->bi_iter.bi_sector = logical >> SECTOR_SHIFT;
- for (int i = 0; i < nr_steps; i++) {
+ while (cur < sectorsize) {
+ struct page *page = bio_iter_page(&failed_bbio->bio, iter);
+ const u32 pg_off = bio_iter_offset(&failed_bbio->bio, iter);
+ const u32 cur_len = min(bio_iter_len(&failed_bbio->bio, iter),
+ sectorsize - cur);
int ret;
- ASSERT(offset_in_page(paddrs[i]) + step <= PAGE_SIZE);
+ ret = bio_add_page(repair_bio, page, cur_len, pg_off);
+ ASSERT(ret == cur_len);
- ret = bio_add_page(repair_bio, phys_to_page(paddrs[i]), step,
- offset_in_page(paddrs[i]));
- ASSERT(ret == step);
+ bio_advance_iter_single(&failed_bbio->bio, &iter, cur_len);
+ cur += cur_len;
}
repair_bbio = btrfs_bio(repair_bio);
@@ -305,18 +277,16 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de
struct btrfs_inode *inode = bbio->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
const u32 sectorsize = fs_info->sectorsize;
- const u32 step = min(sectorsize, PAGE_SIZE);
- const u32 nr_steps = sectorsize / step;
- struct bvec_iter *iter = &bbio->saved_iter;
+ struct bvec_iter iter;
blk_status_t status = bbio->bio.bi_status;
struct btrfs_failed_bio *fbio = NULL;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
- phys_addr_t paddr;
- u32 offset = 0;
/* Read-repair requires the inode field to be set by the submitter. */
ASSERT(inode);
+ /* The original bbio should be sectorsize aligned. */
+ ASSERT(IS_ALIGNED(bbio->saved_iter.bi_size, sectorsize));
+
/*
* Hand off repair bios to the repair code as there is no upper level
* submitter for them.
@@ -329,16 +299,10 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de
/* Clear the I/O error. A failed repair will reset it. */
bbio->bio.bi_status = BLK_STS_OK;
- btrfs_bio_for_each_block(paddr, &bbio->bio, iter, step) {
- paddrs[(offset / step) % nr_steps] = paddr;
- offset += step;
-
- if (IS_ALIGNED(offset, sectorsize)) {
- if (status ||
- !btrfs_data_csum_ok(bbio, dev, offset - sectorsize, paddrs))
- fbio = repair_one_sector(bbio, offset - sectorsize,
- paddrs, fbio);
- }
+ for (iter = bbio->saved_iter; iter.bi_size;
+ bio_advance_iter(&bbio->bio, &iter, sectorsize)) {
+ if (status || !btrfs_bio_data_csum_ok(bbio, &iter, dev))
+ fbio = repair_one_sector(bbio, &iter, fbio);
}
if (bbio->csum != bbio->csum_inline)
kvfree(bbio->csum);
@@ -386,7 +350,7 @@ static void simple_end_io_work(struct work_struct *work)
if (bio_op(bio) == REQ_OP_READ) {
/* Metadata reads are checked and repaired by the submitter. */
- if (is_data_bbio(bbio))
+ if (is_data_inode(bbio->inode))
return btrfs_check_read_bio(bbio, bbio->bio.bi_private);
return btrfs_bio_end_io(bbio, bbio->bio.bi_status);
}
@@ -420,7 +384,7 @@ static void btrfs_raid56_end_io(struct bio *bio)
btrfs_bio_counter_dec(bioc->fs_info);
bbio->mirror_num = bioc->mirror_num;
- if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio))
+ if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode))
btrfs_check_read_bio(bbio, NULL);
else
btrfs_bio_end_io(bbio, bbio->bio.bi_status);
@@ -783,7 +747,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num)
* our bio to the physical disk location, so we need to save the
* original bytenr so we know what we're checksumming.
*/
- if (bio_op(bio) == REQ_OP_WRITE && is_data_bbio(bbio))
+ if (bio_op(bio) == REQ_OP_WRITE && is_data_inode(bbio->inode))
bbio->orig_logical = logical;
bbio->can_use_append = btrfs_use_zone_append(bbio);
@@ -809,7 +773,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num)
* Save the iter for the end_io handler and preload the checksums for
* data reads.
*/
- if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) {
+ if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) {
bbio->saved_iter = bio->bi_iter;
ret = btrfs_lookup_bio_sums(bbio);
status = errno_to_blk_status(ret);
@@ -818,7 +782,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num)
}
if (btrfs_op(bio) == BTRFS_MAP_WRITE) {
- if (is_data_bbio(bbio) && bioc && bioc->use_rst) {
+ if (is_data_inode(bbio->inode) && bioc && bioc->use_rst) {
/*
* No locking for the list update, as we only add to
* the list in the I/O submission path, and list
@@ -925,21 +889,23 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num)
* The I/O is issued synchronously to block the repair read completion from
* freeing the bio.
*
- * @ino: Offending inode number
- * @fileoff: File offset inside the inode
+ * @bbio: Original bbio where the repair is needed
+ * @orig_iter: Points to where the repair starts
* @length: Length of the repair write
- * @logical: Logical address of the range
- * @paddrs: Physical address array of the content
- * @step: Length of for each paddrs
* @mirror_num: Mirror number to write to. Must not be zero
*/
-int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff,
- u32 length, u64 logical, const phys_addr_t paddrs[],
- unsigned int step, int mirror_num)
+int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter,
+ u32 length, int mirror_num)
{
- const u32 nr_steps = DIV_ROUND_UP_POW2(length, step);
+ struct btrfs_inode *inode = bbio->inode;
+ struct btrfs_fs_info *fs_info = inode->root->fs_info;
struct btrfs_io_stripe smap = { 0 };
- struct bio *bio = NULL;
+ struct bvec_iter iter = *orig_iter;
+ struct bio *repair_bio = NULL;
+ const u64 logical = iter.bi_sector << SECTOR_SHIFT;
+ const u64 fileoff = bbio->file_offset +
+ ((iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT);
+ u32 cur = 0;
int ret = 0;
BUG_ON(!mirror_num);
@@ -950,8 +916,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff,
ASSERT(IS_ALIGNED(fileoff, fs_info->sectorsize));
/* Either it's a single data or metadata block. */
ASSERT(length <= BTRFS_MAX_BLOCKSIZE);
- ASSERT(step <= length);
- ASSERT(is_power_of_2(step));
+
+ /* Our current iter should not be before the original bbio saved_iter. */
+ ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector);
/*
* The fs either mounted RO or hit critical errors, no need
@@ -979,15 +946,22 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff,
goto out_counter_dec;
}
- bio = bio_alloc(smap.dev->bdev, nr_steps, REQ_OP_WRITE | REQ_SYNC, GFP_NOFS);
- bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT;
- for (int i = 0; i < nr_steps; i++) {
- ret = bio_add_page(bio, phys_to_page(paddrs[i]), step, offset_in_page(paddrs[i]));
- /* We should have allocated enough slots to contain all the different pages. */
- ASSERT(ret == step);
+ repair_bio = bio_alloc(smap.dev->bdev, max(1, length >> PAGE_SHIFT),
+ REQ_OP_WRITE | REQ_SYNC, GFP_NOFS);
+ repair_bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT;
+ while (cur < length) {
+ struct page *page = bio_iter_page(&bbio->bio, iter);
+ const u32 pg_off = bio_iter_offset(&bbio->bio, iter);
+ const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), length - cur);
+
+ ret = bio_add_page(repair_bio, page, cur_len, pg_off);
+ ASSERT(ret == cur_len);
+ bio_advance_iter_single(&bbio->bio, &iter, cur_len);
+ cur += cur_len;
}
- ret = submit_bio_wait(bio);
- bio_put(bio);
+
+ ret = submit_bio_wait(repair_bio);
+ bio_put(repair_bio);
if (ret) {
/* try to remap that extent elsewhere? */
btrfs_dev_stat_inc_and_print(smap.dev, BTRFS_DEV_STAT_WRITE_ERRS);
@@ -995,8 +969,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff,
}
btrfs_info_rl(fs_info,
- "read error corrected: ino %llu off %llu (dev %s sector %llu)",
- ino, fileoff, btrfs_dev_name(smap.dev),
+ "read error corrected: root %llu ino %llu off %llu (dev %s sector %llu)",
+ btrfs_root_id(inode->root), btrfs_ino(inode), fileoff,
+ btrfs_dev_name(smap.dev),
smap.physical >> SECTOR_SHIFT);
ret = 0;
diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h
index 303ed6c7103d..bbf362b8668b 100644
--- a/fs/btrfs/bio.h
+++ b/fs/btrfs/bio.h
@@ -58,7 +58,6 @@ struct btrfs_bio {
struct btrfs_ordered_extent *ordered;
struct btrfs_ordered_sum *sums;
struct work_struct csum_work;
- struct completion csum_done;
struct bvec_iter csum_saved_iter;
u64 orig_physical;
u64 orig_logical;
@@ -93,9 +92,6 @@ struct btrfs_bio {
/* Whether the bio is coming from copy_remapped_data_io(). */
bool is_remap:1;
- /* Whether the csum generation for data write is async. */
- bool async_csum:1;
-
/* Whether the bio is written using zone append. */
bool can_use_append:1;
@@ -126,8 +122,7 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status);
void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num);
void btrfs_submit_repair_write(struct btrfs_bio *bbio, int mirror_num, bool dev_replace);
-int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff,
- u32 length, u64 logical, const phys_addr_t paddrs[],
- unsigned int step, int mirror_num);
+int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter,
+ u32 length, int mirror_num);
#endif
diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c
index ee182369254c..e6080cb47c89 100644
--- a/fs/btrfs/block-group.c
+++ b/fs/btrfs/block-group.c
@@ -904,22 +904,6 @@ static noinline void caching_thread(struct btrfs_work *work)
down_read(&fs_info->commit_root_sem);
load_block_group_size_class(caching_ctl);
- if (btrfs_test_opt(fs_info, SPACE_CACHE)) {
- ret = load_free_space_cache(block_group);
- if (ret == 1) {
- ret = 0;
- goto done;
- }
-
- /*
- * We failed to load the space cache, set ourselves to
- * CACHE_STARTED and carry on.
- */
- spin_lock(&block_group->lock);
- block_group->cached = BTRFS_CACHE_STARTED;
- spin_unlock(&block_group->lock);
- wake_up(&caching_ctl->wait);
- }
/*
* If we are in the transaction that populated the free space tree we
@@ -933,7 +917,7 @@ static noinline void caching_thread(struct btrfs_work *work)
ret = btrfs_load_free_space_tree(caching_ctl);
else
ret = load_extent_tree_free(caching_ctl);
-done:
+
spin_lock(&block_group->lock);
block_group->caching_ctl = NULL;
block_group->cached = ret ? BTRFS_CACHE_ERROR : BTRFS_CACHE_FINISHED;
@@ -1194,36 +1178,21 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans,
goto out;
}
- /*
- * get the inode first so any iput calls done for the io_list
- * aren't the final iput (no unlinks allowed now)
- */
inode = lookup_free_space_inode(block_group, path);
- mutex_lock(&trans->transaction->cache_write_mutex);
/*
- * Make sure our free space cache IO is done before removing the
- * free space inode
+ * Do not delete the block group item while
+ * btrfs_start_dirty_block_groups() is updating it.
*/
+ mutex_lock(&trans->transaction->dirty_bgs_update_mutex);
spin_lock(&trans->transaction->dirty_bgs_lock);
- if (!list_empty(&block_group->io_list)) {
- list_del_init(&block_group->io_list);
-
- WARN_ON(!IS_ERR(inode) && inode != block_group->io_ctl.inode);
-
- spin_unlock(&trans->transaction->dirty_bgs_lock);
- btrfs_wait_cache_io(trans, block_group, path);
- btrfs_put_block_group(block_group);
- spin_lock(&trans->transaction->dirty_bgs_lock);
- }
-
if (!list_empty(&block_group->dirty_list)) {
list_del_init(&block_group->dirty_list);
remove_rsv = true;
btrfs_put_block_group(block_group);
}
spin_unlock(&trans->transaction->dirty_bgs_lock);
- mutex_unlock(&trans->transaction->cache_write_mutex);
+ mutex_unlock(&trans->transaction->dirty_bgs_update_mutex);
ret = btrfs_remove_free_space_inode(trans, inode, block_group);
if (unlikely(ret)) {
@@ -1287,7 +1256,6 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans,
spin_lock(&trans->transaction->dirty_bgs_lock);
WARN_ON(!list_empty(&block_group->dirty_list));
- WARN_ON(!list_empty(&block_group->io_list));
spin_unlock(&trans->transaction->dirty_bgs_lock);
btrfs_remove_free_space_cache(block_group);
@@ -1612,7 +1580,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info)
space_info = block_group->space_info;
- if (ret || btrfs_mixed_space_info(space_info)) {
+ if (btrfs_mixed_space_info(space_info)) {
btrfs_put_block_group(block_group);
continue;
}
@@ -1727,6 +1695,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info)
ret = inc_block_group_ro(block_group, false);
up_write(&space_info->groups_sem);
if (ret < 0) {
+ btrfs_link_bg_list(block_group, &retry_list);
ret = 0;
goto next;
}
@@ -1749,6 +1718,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info)
block_group->start);
if (IS_ERR(trans)) {
btrfs_dec_block_group_ro(block_group);
+ btrfs_link_bg_list(block_group, &retry_list);
ret = PTR_ERR(trans);
goto next;
}
@@ -1759,6 +1729,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info)
*/
if (!clean_pinned_extents(trans, block_group)) {
btrfs_dec_block_group_ro(block_group);
+ btrfs_link_bg_list(block_group, &retry_list);
goto end_trans;
}
@@ -1845,6 +1816,8 @@ end_trans:
next:
btrfs_put_block_group(block_group);
spin_lock(&fs_info->unused_bgs_lock);
+ if (ret)
+ break;
}
list_splice_tail(&retry_list, &fs_info->unused_bgs);
spin_unlock(&fs_info->unused_bgs_lock);
@@ -2431,7 +2404,6 @@ static struct btrfs_block_group *btrfs_create_block_group(
INIT_LIST_HEAD(&cache->ro_list);
INIT_LIST_HEAD(&cache->discard_list);
INIT_LIST_HEAD(&cache->dirty_list);
- INIT_LIST_HEAD(&cache->io_list);
INIT_LIST_HEAD(&cache->active_bg_list);
btrfs_init_free_space_ctl(cache, cache->free_space_ctl);
atomic_set(&cache->frozen, 0);
@@ -2487,8 +2459,7 @@ static int check_chunk_block_group_mappings(struct btrfs_fs_info *fs_info)
static int read_one_block_group(struct btrfs_fs_info *info,
struct btrfs_block_group_item_v2 *bgi,
- const struct btrfs_key *key,
- bool need_clear)
+ const struct btrfs_key *key)
{
struct btrfs_block_group *cache;
const bool mixed = btrfs_fs_incompat(info, MIXED_GROUPS);
@@ -2514,20 +2485,6 @@ static int read_one_block_group(struct btrfs_fs_info *info,
btrfs_set_free_space_tree_thresholds(cache);
- if (need_clear) {
- /*
- * When we mount with old space cache, we need to
- * set BTRFS_DC_CLEAR and set dirty flag.
- *
- * a) Setting 'BTRFS_DC_CLEAR' makes sure that we
- * truncate the old free space cache inode and
- * setup a new one.
- * b) Setting 'dirty flag' makes sure that we flush
- * the new space cache info onto disk.
- */
- if (btrfs_test_opt(info, SPACE_CACHE))
- cache->disk_cache_state = BTRFS_DC_CLEAR;
- }
if (!mixed && ((cache->flags & BTRFS_BLOCK_GROUP_METADATA) &&
(cache->flags & BTRFS_BLOCK_GROUP_DATA))) {
btrfs_err(info,
@@ -2668,8 +2625,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info)
struct btrfs_block_group *cache;
struct btrfs_space_info *space_info;
struct btrfs_key key;
- bool need_clear = false;
- u64 cache_gen;
/*
* Either no extent root (with ibadroots rescue option) or we have
@@ -2690,13 +2645,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info)
if (!path)
return -ENOMEM;
- cache_gen = btrfs_super_cache_generation(info->super_copy);
- if (btrfs_test_opt(info, SPACE_CACHE) &&
- btrfs_super_generation(info->super_copy) != cache_gen)
- need_clear = true;
- if (btrfs_test_opt(info, CLEAR_CACHE))
- need_clear = true;
-
while (1) {
struct btrfs_block_group_item_v2 bgi;
struct extent_buffer *leaf;
@@ -2725,7 +2673,7 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info)
btrfs_item_key_to_cpu(leaf, &key, slot);
btrfs_release_path(path);
- ret = read_one_block_group(info, &bgi, &key, need_clear);
+ ret = read_one_block_group(info, &bgi, &key);
if (ret < 0)
goto error;
key.objectid += key.offset;
@@ -3373,207 +3321,16 @@ fail:
}
-static void cache_save_setup(struct btrfs_block_group *block_group,
- struct btrfs_trans_handle *trans,
- struct btrfs_path *path)
-{
- struct btrfs_fs_info *fs_info = block_group->fs_info;
- struct inode *inode = NULL;
- struct extent_changeset *data_reserved = NULL;
- u64 alloc_hint = 0;
- int dcs = BTRFS_DC_ERROR;
- u64 cache_size = 0;
- int retries = 0;
- int ret = 0;
-
- if (!btrfs_test_opt(fs_info, SPACE_CACHE))
- return;
-
- /*
- * If this block group is smaller than 100 megs don't bother caching the
- * block group.
- */
- if (block_group->length < (100 * SZ_1M)) {
- spin_lock(&block_group->lock);
- block_group->disk_cache_state = BTRFS_DC_WRITTEN;
- spin_unlock(&block_group->lock);
- return;
- }
-
- if (TRANS_ABORTED(trans))
- return;
-again:
- inode = lookup_free_space_inode(block_group, path);
- if (IS_ERR(inode) && PTR_ERR(inode) != -ENOENT) {
- ret = PTR_ERR(inode);
- btrfs_release_path(path);
- goto out;
- }
-
- if (IS_ERR(inode)) {
- if (retries) {
- ret = PTR_ERR(inode);
- btrfs_err(fs_info,
- "failed to lookup free space inode after creation for block group %llu: %d",
- block_group->start, ret);
- goto out_free;
- }
- retries++;
-
- if (block_group->ro)
- goto out_free;
-
- ret = create_free_space_inode(trans, block_group, path);
- if (ret)
- goto out_free;
- goto again;
- }
-
- /*
- * We want to set the generation to 0, that way if anything goes wrong
- * from here on out we know not to trust this cache when we load up next
- * time.
- */
- BTRFS_I(inode)->generation = 0;
- ret = btrfs_update_inode(trans, BTRFS_I(inode));
- if (unlikely(ret)) {
- /*
- * So theoretically we could recover from this, simply set the
- * super cache generation to 0 so we know to invalidate the
- * cache, but then we'd have to keep track of the block groups
- * that fail this way so we know we _have_ to reset this cache
- * before the next commit or risk reading stale cache. So to
- * limit our exposure to horrible edge cases lets just abort the
- * transaction, this only happens in really bad situations
- * anyway.
- */
- btrfs_abort_transaction(trans, ret);
- goto out_put;
- }
-
- /* We've already setup this transaction, go ahead and exit */
- if (block_group->cache_generation == trans->transid &&
- i_size_read(inode)) {
- dcs = BTRFS_DC_SETUP;
- goto out_put;
- }
-
- if (i_size_read(inode) > 0) {
- ret = btrfs_check_trunc_cache_free_space(fs_info,
- &fs_info->global_block_rsv);
- if (ret)
- goto out_put;
-
- ret = btrfs_truncate_free_space_cache(trans, NULL, inode);
- if (ret)
- goto out_put;
- }
-
- spin_lock(&block_group->lock);
- if (block_group->cached != BTRFS_CACHE_FINISHED ||
- !btrfs_test_opt(fs_info, SPACE_CACHE)) {
- /*
- * don't bother trying to write stuff out _if_
- * a) we're not cached,
- * b) we're with nospace_cache mount option,
- * c) we're with v2 space_cache (FREE_SPACE_TREE).
- */
- dcs = BTRFS_DC_WRITTEN;
- spin_unlock(&block_group->lock);
- goto out_put;
- }
- spin_unlock(&block_group->lock);
-
- /*
- * We hit an ENOSPC when setting up the cache in this transaction, just
- * skip doing the setup, we've already cleared the cache so we're safe.
- */
- if (test_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags))
- goto out_put;
-
- /*
- * Try to preallocate enough space based on how big the block group is.
- * Keep in mind this has to include any pinned space which could end up
- * taking up quite a bit since it's not folded into the other space
- * cache.
- */
- cache_size = div_u64(block_group->length, SZ_256M);
- if (!cache_size)
- cache_size = 1;
-
- cache_size *= 16;
- cache_size *= fs_info->sectorsize;
-
- ret = btrfs_check_data_free_space(BTRFS_I(inode), &data_reserved, 0,
- cache_size, false);
- if (ret)
- goto out_put;
-
- ret = btrfs_prealloc_file_range_trans(inode, trans, 0, 0, cache_size,
- cache_size, cache_size,
- &alloc_hint);
- /*
- * Our cache requires contiguous chunks so that we don't modify a bunch
- * of metadata or split extents when writing the cache out, which means
- * we can enospc if we are heavily fragmented in addition to just normal
- * out of space conditions. So if we hit this just skip setting up any
- * other block groups for this transaction, maybe we'll unpin enough
- * space the next time around.
- */
- if (!ret)
- dcs = BTRFS_DC_SETUP;
- else if (ret == -ENOSPC)
- set_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags);
-
-out_put:
- iput(inode);
-out_free:
- btrfs_release_path(path);
-out:
- spin_lock(&block_group->lock);
- if (!ret && dcs == BTRFS_DC_SETUP)
- block_group->cache_generation = trans->transid;
- block_group->disk_cache_state = dcs;
- spin_unlock(&block_group->lock);
-
- extent_changeset_free(data_reserved);
-}
-
-int btrfs_setup_space_cache(struct btrfs_trans_handle *trans)
-{
- struct btrfs_fs_info *fs_info = trans->fs_info;
- struct btrfs_block_group *cache, *tmp;
- struct btrfs_transaction *cur_trans = trans->transaction;
- BTRFS_PATH_AUTO_FREE(path);
-
- if (list_empty(&cur_trans->dirty_bgs) ||
- !btrfs_test_opt(fs_info, SPACE_CACHE))
- return 0;
-
- path = btrfs_alloc_path();
- if (!path)
- return -ENOMEM;
-
- /* Could add new block groups, use _safe just in case */
- list_for_each_entry_safe(cache, tmp, &cur_trans->dirty_bgs,
- dirty_list) {
- if (cache->disk_cache_state == BTRFS_DC_CLEAR)
- cache_save_setup(cache, trans, path);
- }
-
- return 0;
-}
-
/*
- * Transaction commit does final block group cache writeback during a critical
+ * Transaction commit does the final block group item updates during a critical
* section where nothing is allowed to change the FS. This is required in
- * order for the cache to actually match the block group, but can introduce a
+ * order for the items to actually match the block groups, but can introduce a
* lot of latency into the commit.
*
- * So, btrfs_start_dirty_block_groups is here to kick off block group cache IO.
- * There's a chance we'll have to redo some of it if the block group changes
- * again during the commit, but it greatly reduces the commit latency by
- * getting rid of the easy block groups while we're still allowing others to
+ * So, btrfs_start_dirty_block_groups is here to update the block group items
+ * early. There's a chance we'll have to redo some of it if the block group
+ * changes again during the commit, but it greatly reduces the commit latency
+ * by getting rid of the easy block groups while we're still allowing others to
* join the commit.
*/
int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans)
@@ -3582,10 +3339,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans)
struct btrfs_block_group *cache;
struct btrfs_transaction *cur_trans = trans->transaction;
int ret = 0;
- int should_put;
BTRFS_PATH_AUTO_FREE(path);
LIST_HEAD(dirty);
- struct list_head *io = &cur_trans->io_bgs;
int loops = 0;
spin_lock(&cur_trans->dirty_bgs_lock);
@@ -3609,33 +3364,18 @@ again:
}
/*
- * cache_write_mutex is here only to save us from balance or automatic
- * removal of empty block groups deleting this block group while we are
- * writing out the cache
+ * dirty_bgs_update_mutex is here only to save us from balance or
+ * automatic removal of empty block groups deleting this block group
+ * while we are updating its item
*/
- mutex_lock(&trans->transaction->cache_write_mutex);
+ mutex_lock(&trans->transaction->dirty_bgs_update_mutex);
while (!list_empty(&dirty)) {
bool drop_reserve = true;
cache = list_first_entry(&dirty, struct btrfs_block_group,
dirty_list);
- /*
- * This can happen if something re-dirties a block group that
- * is already under IO. Just wait for it to finish and then do
- * it all again
- */
- if (!list_empty(&cache->io_list)) {
- list_del_init(&cache->io_list);
- btrfs_wait_cache_io(trans, cache, path);
- btrfs_put_block_group(cache);
- }
-
/*
- * btrfs_wait_cache_io uses the cache->dirty_list to decide if
- * it should update the cache_state. Don't delete until after
- * we wait.
- *
* Since we're not running in the commit critical section
* we need the dirty_bgs_lock to protect from update_block_group
*/
@@ -3643,72 +3383,39 @@ again:
list_del_init(&cache->dirty_list);
spin_unlock(&cur_trans->dirty_bgs_lock);
- should_put = 1;
-
- cache_save_setup(cache, trans, path);
-
- if (cache->disk_cache_state == BTRFS_DC_SETUP) {
- cache->io_ctl.inode = NULL;
- ret = btrfs_write_out_cache(trans, cache, path);
- if (ret == 0 && cache->io_ctl.inode) {
- should_put = 0;
-
- /*
- * The cache_write_mutex is protecting the
- * io_list, also refer to the definition of
- * btrfs_transaction::io_bgs for more details
- */
- list_add_tail(&cache->io_list, io);
- } else {
- /*
- * If we failed to write the cache, the
- * generation will be bad and life goes on
- */
- ret = 0;
- }
- }
- if (!ret) {
- ret = update_block_group_item(trans, path, cache);
- /*
- * Our block group might still be attached to the list
- * of new block groups in the transaction handle of some
- * other task (struct btrfs_trans_handle->new_bgs). This
- * means its block group item isn't yet in the extent
- * tree. If this happens ignore the error, as we will
- * try again later in the critical section of the
- * transaction commit.
- */
- if (ret == -ENOENT) {
- ret = 0;
- spin_lock(&cur_trans->dirty_bgs_lock);
- if (list_empty(&cache->dirty_list)) {
- list_add_tail(&cache->dirty_list,
- &cur_trans->dirty_bgs);
- btrfs_get_block_group(cache);
- drop_reserve = false;
- }
- spin_unlock(&cur_trans->dirty_bgs_lock);
- } else if (ret) {
- btrfs_abort_transaction(trans, ret);
+ ret = update_block_group_item(trans, path, cache);
+ /*
+ * Our block group might still be attached to the list of new
+ * block groups in the transaction handle of some other task
+ * (struct btrfs_trans_handle->new_bgs). This means its block
+ * group item isn't yet in the extent tree. If this happens
+ * ignore the error, as we will try again later in the critical
+ * section of the transaction commit.
+ */
+ if (ret == -ENOENT) {
+ ret = 0;
+ spin_lock(&cur_trans->dirty_bgs_lock);
+ if (list_empty(&cache->dirty_list)) {
+ list_add_tail(&cache->dirty_list,
+ &cur_trans->dirty_bgs);
+ btrfs_get_block_group(cache);
+ drop_reserve = false;
}
+ spin_unlock(&cur_trans->dirty_bgs_lock);
+ } else if (ret) {
+ btrfs_abort_transaction(trans, ret);
}
- /* If it's not on the io list, we need to put the block group */
- if (should_put)
- btrfs_put_block_group(cache);
+ btrfs_put_block_group(cache);
if (drop_reserve)
btrfs_dec_delayed_refs_rsv_bg_updates(fs_info);
- /*
- * Avoid blocking other tasks for too long. It might even save
- * us from writing caches for block groups that are going to be
- * removed.
- */
- mutex_unlock(&trans->transaction->cache_write_mutex);
+ /* Avoid blocking other tasks for too long. */
+ mutex_unlock(&trans->transaction->dirty_bgs_update_mutex);
if (ret)
goto out;
- mutex_lock(&trans->transaction->cache_write_mutex);
+ mutex_lock(&trans->transaction->dirty_bgs_update_mutex);
}
- mutex_unlock(&trans->transaction->cache_write_mutex);
+ mutex_unlock(&trans->transaction->dirty_bgs_update_mutex);
/*
* Go through delayed refs for all the stuff we've just kicked off
@@ -3722,7 +3429,7 @@ again:
list_splice_init(&cur_trans->dirty_bgs, &dirty);
/*
* dirty_bgs_lock protects us from concurrent block group
- * deletes too (not just cache_write_mutex).
+ * deletes too (not just dirty_bgs_update_mutex).
*/
if (!list_empty(&dirty)) {
spin_unlock(&cur_trans->dirty_bgs_lock);
@@ -3747,121 +3454,34 @@ int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans)
struct btrfs_block_group *cache;
struct btrfs_transaction *cur_trans = trans->transaction;
int ret = 0;
- int should_put;
BTRFS_PATH_AUTO_FREE(path);
- struct list_head *io = &cur_trans->io_bgs;
path = btrfs_alloc_path();
if (!path)
return -ENOMEM;
- /*
- * Even though we are in the critical section of the transaction commit,
- * we can still have concurrent tasks adding elements to this
- * transaction's list of dirty block groups. These tasks correspond to
- * endio free space workers started when writeback finishes for a
- * space cache, which run inode.c:btrfs_finish_ordered_io(), and can
- * allocate new block groups as a result of COWing nodes of the root
- * tree when updating the free space inode. The writeback for the space
- * caches is triggered by an earlier call to
- * btrfs_start_dirty_block_groups() and iterations of the following
- * loop.
- * Also we want to do the cache_save_setup first and then run the
- * delayed refs to make sure we have the best chance at doing this all
- * in one shot.
- */
spin_lock(&cur_trans->dirty_bgs_lock);
while (!list_empty(&cur_trans->dirty_bgs)) {
cache = list_first_entry(&cur_trans->dirty_bgs,
struct btrfs_block_group,
dirty_list);
-
- /*
- * This can happen if cache_save_setup re-dirties a block group
- * that is already under IO. Just wait for it to finish and
- * then do it all again
- */
- if (!list_empty(&cache->io_list)) {
- spin_unlock(&cur_trans->dirty_bgs_lock);
- list_del_init(&cache->io_list);
- btrfs_wait_cache_io(trans, cache, path);
- btrfs_put_block_group(cache);
- spin_lock(&cur_trans->dirty_bgs_lock);
- }
-
- /*
- * Don't remove from the dirty list until after we've waited on
- * any pending IO
- */
list_del_init(&cache->dirty_list);
spin_unlock(&cur_trans->dirty_bgs_lock);
- should_put = 1;
-
- cache_save_setup(cache, trans, path);
if (!ret)
ret = btrfs_run_delayed_refs(trans, U64_MAX);
-
- if (!ret && cache->disk_cache_state == BTRFS_DC_SETUP) {
- cache->io_ctl.inode = NULL;
- ret = btrfs_write_out_cache(trans, cache, path);
- if (ret == 0 && cache->io_ctl.inode) {
- should_put = 0;
- list_add_tail(&cache->io_list, io);
- } else {
- /*
- * If we failed to write the cache, the
- * generation will be bad and life goes on
- */
- ret = 0;
- }
- }
if (!ret) {
ret = update_block_group_item(trans, path, cache);
- /*
- * One of the free space endio workers might have
- * created a new block group while updating a free space
- * cache's inode (at inode.c:btrfs_finish_ordered_io())
- * and hasn't released its transaction handle yet, in
- * which case the new block group is still attached to
- * its transaction handle and its creation has not
- * finished yet (no block group item in the extent tree
- * yet, etc). If this is the case, wait for all free
- * space endio workers to finish and retry. This is a
- * very rare case so no need for a more efficient and
- * complex approach.
- */
- if (ret == -ENOENT) {
- wait_event(cur_trans->writer_wait,
- atomic_read(&cur_trans->num_writers) == 1);
- ret = update_block_group_item(trans, path, cache);
- if (ret)
- btrfs_abort_transaction(trans, ret);
- } else if (ret) {
+ if (ret)
btrfs_abort_transaction(trans, ret);
- }
}
- /* If its not on the io list, we need to put the block group */
- if (should_put)
- btrfs_put_block_group(cache);
+ btrfs_put_block_group(cache);
btrfs_dec_delayed_refs_rsv_bg_updates(fs_info);
spin_lock(&cur_trans->dirty_bgs_lock);
}
spin_unlock(&cur_trans->dirty_bgs_lock);
- /*
- * Refer to the definition of io_bgs member for details why it's safe
- * to use it without any locking
- */
- while (!list_empty(io)) {
- cache = list_first_entry(io, struct btrfs_block_group,
- io_list);
- list_del_init(&cache->io_list);
- btrfs_wait_cache_io(trans, cache, path);
- btrfs_put_block_group(cache);
- }
-
return ret;
}
@@ -3905,10 +3525,10 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans,
factor = btrfs_bg_type_to_factor(cache->flags);
/*
- * If this block group has free space cache written out, we need to make
- * sure to load it if we are removing space. This is because we need
- * the unpinning stage to actually add the space back to the block group,
- * otherwise we will leak space.
+ * Make sure the free space of this block group is loaded if we are
+ * removing space. This is because we need the unpinning stage to
+ * actually add the space back to the block group, otherwise we will
+ * leak space.
*/
if (!alloc && !btrfs_block_group_done(cache))
btrfs_cache_block_group(cache, true);
@@ -3916,10 +3536,6 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans,
spin_lock(&space_info->lock);
spin_lock(&cache->lock);
- if (btrfs_test_opt(info, SPACE_CACHE) &&
- cache->disk_cache_state < BTRFS_DC_CLEAR)
- cache->disk_cache_state = BTRFS_DC_CLEAR;
-
old_val = cache->used;
if (alloc) {
old_val += num_bytes;
@@ -3964,7 +3580,7 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans,
/*
* No longer have used bytes in this block group, queue it for deletion.
* We do this after adding the block group to the dirty list to avoid
- * races between cleaner kthread and space cache writeout.
+ * races between the cleaner kthread and the dirty block group writeout.
*/
if (!alloc && old_val == 0) {
if (!btrfs_test_opt(info, DISCARD_ASYNC))
@@ -4640,7 +4256,6 @@ void btrfs_put_block_group_cache(struct btrfs_fs_info *info)
block_group->inode = NULL;
spin_unlock(&block_group->lock);
- ASSERT(block_group->io_ctl.inode == NULL);
iput(&inode->vfs_inode);
} else {
spin_unlock(&block_group->lock);
@@ -4777,7 +4392,6 @@ int btrfs_free_block_groups(struct btrfs_fs_info *info)
btrfs_remove_free_space_cache(block_group);
ASSERT(block_group->cached != BTRFS_CACHE_STARTED);
ASSERT(list_empty(&block_group->dirty_list));
- ASSERT(list_empty(&block_group->io_list));
ASSERT(list_empty(&block_group->bg_list));
ASSERT(refcount_read(&block_group->refs) == 1);
ASSERT(block_group->swap_extents == 0);
diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h
index 69d56864d4ba..d567ed822e55 100644
--- a/fs/btrfs/block-group.h
+++ b/fs/btrfs/block-group.h
@@ -20,13 +20,6 @@ struct btrfs_fs_info;
struct btrfs_inode;
struct btrfs_trans_handle;
-enum btrfs_disk_cache_state {
- BTRFS_DC_WRITTEN,
- BTRFS_DC_ERROR,
- BTRFS_DC_CLEAR,
- BTRFS_DC_SETUP,
-};
-
enum btrfs_block_group_size_class {
/* Unset */
BTRFS_BG_SZ_NONE,
@@ -131,11 +124,10 @@ struct btrfs_block_group {
u64 delalloc_bytes;
u64 bytes_super;
u64 flags;
- u64 cache_generation;
u64 global_root_id;
u64 remap_bytes;
u32 identity_remap_count;
- /* The last commited identity_remap_count value of this block group. */
+ /* The last committed identity_remap_count value of this block group. */
u32 last_identity_remap_count;
/*
* The last committed used bytes of this block group, if the above @used
@@ -171,8 +163,6 @@ struct btrfs_block_group {
unsigned long full_stripe_len;
unsigned long runtime_flags;
- enum btrfs_disk_cache_state disk_cache_state;
-
/* Cache tracking stuff */
enum btrfs_caching_type cached;
struct btrfs_caching_control *caching_ctl;
@@ -228,9 +218,6 @@ struct btrfs_block_group {
/* For dirty block groups */
struct list_head dirty_list;
- struct list_head io_list;
-
- struct btrfs_io_ctl io_ctl;
/*
* Incremented when doing extent allocations and holding a read lock
@@ -368,7 +355,6 @@ int btrfs_inc_block_group_ro(struct btrfs_block_group *cache,
void btrfs_dec_block_group_ro(struct btrfs_block_group *cache);
int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans);
int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans);
-int btrfs_setup_space_cache(struct btrfs_trans_handle *trans);
int btrfs_update_block_group(struct btrfs_trans_handle *trans,
u64 bytenr, u64 num_bytes, bool alloc);
int btrfs_add_reserved_bytes(struct btrfs_block_group *cache,
diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h
index 1082fa92c145..114a5c38afd3 100644
--- a/fs/btrfs/btrfs_inode.h
+++ b/fs/btrfs/btrfs_inode.h
@@ -32,6 +32,7 @@ struct btrfs_trans_handle;
struct btrfs_bio;
struct btrfs_file_extent;
struct btrfs_delayed_node;
+struct btrfs_dir_index_prealloc;
/*
* Since we search a directory based on f_pos (struct dir_context::pos) we have
@@ -402,8 +403,6 @@ static inline void btrfs_mod_outstanding_extents(struct btrfs_inode *inode,
{
lockdep_assert_held(&inode->lock);
inode->outstanding_extents += mod;
- if (btrfs_is_free_space_inode(inode))
- return;
trace_btrfs_inode_mod_outstanding_extents(inode->root, btrfs_ino(inode),
mod, inode->outstanding_extents);
}
@@ -507,14 +506,10 @@ static inline void btrfs_set_inode_mapping_order(struct btrfs_inode *inode)
inode->root->fs_info->block_max_order);
}
-void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info,
- const phys_addr_t paddr, u8 *dest);
-void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info,
- const phys_addr_t paddrs[], u8 *dest);
-int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum,
- const u8 * const csum_expected);
-bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev,
- u32 bio_offset, const phys_addr_t paddrs[]);
+bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter,
+ struct btrfs_device *dev);
+void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio,
+ const struct bvec_iter *orig_iter, u8 *csum);
noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len,
struct btrfs_file_extent *file_extent,
bool nowait);
@@ -527,7 +522,8 @@ int btrfs_unlink_inode(struct btrfs_trans_handle *trans,
const struct fscrypt_str *name);
int btrfs_add_link(struct btrfs_trans_handle *trans,
struct btrfs_inode *parent_inode, struct btrfs_inode *inode,
- const struct fscrypt_str *name, bool add_backref, u64 index);
+ const struct fscrypt_str *name, bool add_backref, u64 index,
+ struct btrfs_dir_index_prealloc *prealloc);
int btrfs_delete_subvolume(struct btrfs_inode *dir, struct dentry *dentry);
int btrfs_truncate_block(struct btrfs_inode *inode, u64 offset, u64 start, u64 end);
@@ -594,10 +590,6 @@ int btrfs_wait_on_delayed_iputs(struct btrfs_fs_info *fs_info);
int btrfs_prealloc_file_range(struct inode *inode, int mode,
u64 start, u64 num_bytes, u64 min_size,
loff_t actual_len, u64 *alloc_hint);
-int btrfs_prealloc_file_range_trans(struct inode *inode,
- struct btrfs_trans_handle *trans, int mode,
- u64 start, u64 num_bytes, u64 min_size,
- loff_t actual_len, u64 *alloc_hint);
int btrfs_run_delalloc_range(struct btrfs_inode *inode, struct folio *locked_folio,
u64 start, u64 end, struct writeback_control *wbc);
void btrfs_queue_writepage_fixup(struct btrfs_inode *inode, struct folio *folio);
diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c
index c62b5148d5ac..ad90e14032db 100644
--- a/fs/btrfs/compression.c
+++ b/fs/btrfs/compression.c
@@ -168,10 +168,10 @@ static unsigned long btrfs_compr_pool_scan(struct shrinker *sh, struct shrink_co
spin_unlock(&compr_pool.lock);
list_for_each_safe(tmp, next, &remove) {
- struct page *page = list_entry(tmp, struct page, lru);
+ struct folio *folio = list_entry(tmp, struct folio, lru);
- ASSERT(page_ref_count(page) == 1);
- put_page(page);
+ ASSERT(folio_ref_count(folio) == 1);
+ folio_put(folio);
}
return freed;
@@ -431,7 +431,7 @@ static noinline int add_ra_bio_folios(struct inode *inode, u64 compressed_end,
}
/*
- * Since add_ra_bio_pages() is always speculative, suppress
+ * Since add_ra_bio_folios() is always speculative, suppress
* allocation warnings.
*/
masked_constraint_gfp = mapping_gfp_constraint(mapping, constraint_gfp);
@@ -960,7 +960,7 @@ bool btrfs_compress_level_valid(unsigned int type, int level)
return levels->min_level <= level && level <= levels->max_level;
}
-/* Wrapper around find_get_page(), with extra error message. */
+/* Wrapper around filemap_get_folio(), with extra error message. */
int btrfs_compress_filemap_get_folio(struct address_space *mapping, u64 start,
struct folio **in_folio_ret)
{
@@ -1488,10 +1488,11 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws)
static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end,
struct heuristic_ws *ws)
{
- struct page *page;
- pgoff_t index, index_end;
- u32 i, curr_sample_pos;
- u8 *in_data;
+ const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize;
+ u64 cur = start;
+ u32 curr_sample_pos = 0;
+
+ ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize));
/*
* Compression handles the input data by chunks of 128KiB
@@ -1502,38 +1503,30 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end,
* MAX_SAMPLE_SIZE - calculated under assumption that heuristic will
* process no more than BTRFS_MAX_UNCOMPRESSED at a time.
*/
- if (end - start > BTRFS_MAX_UNCOMPRESSED)
- end = start + BTRFS_MAX_UNCOMPRESSED;
-
- index = start >> PAGE_SHIFT;
- index_end = end >> PAGE_SHIFT;
-
- /* Don't miss unaligned end */
- if (!PAGE_ALIGNED(end))
- index_end++;
-
- curr_sample_pos = 0;
- while (index < index_end) {
- page = find_get_page(inode->i_mapping, index);
- in_data = kmap_local_page(page);
- /* Handle case where the start is not aligned to PAGE_SIZE */
- i = start % PAGE_SIZE;
- while (i < PAGE_SIZE - SAMPLING_READ_SIZE) {
- /* Don't sample any garbage from the last page */
- if (start > end - SAMPLING_READ_SIZE)
- break;
- memcpy(&ws->sample[curr_sample_pos], &in_data[i],
- SAMPLING_READ_SIZE);
- i += SAMPLING_INTERVAL;
- start += SAMPLING_INTERVAL;
+ if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED)
+ end = start + BTRFS_MAX_UNCOMPRESSED - 1;
+
+ while (cur < end) {
+ struct folio *folio;
+ void *in_data;
+ u64 next_pos;
+
+ folio = filemap_get_folio(inode->i_mapping, cur >> PAGE_SHIFT);
+ /* All folios inside the range should exist and be locked. */
+ ASSERT(!IS_ERR(folio));
+ next_pos = min_t(u64, end + 1, folio_next_pos(folio));
+ in_data = kmap_local_folio(folio, 0);
+
+ for (; cur < next_pos; cur += SAMPLING_INTERVAL) {
+ memcpy(&ws->sample[curr_sample_pos],
+ in_data + offset_in_folio(folio, cur),
+ SAMPLING_READ_SIZE);
curr_sample_pos += SAMPLING_READ_SIZE;
}
kunmap_local(in_data);
- put_page(page);
-
- index++;
+ folio_put(folio);
+ cur = next_pos;
}
-
ws->sample_size = curr_sample_pos;
}
diff --git a/fs/btrfs/ctree.c b/fs/btrfs/ctree.c
index 8fe330d81b8f..fef0e49dd918 100644
--- a/fs/btrfs/ctree.c
+++ b/fs/btrfs/ctree.c
@@ -3943,7 +3943,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans,
orig_offset = btrfs_item_offset(leaf, path->slots[0]);
item_size = btrfs_item_size(leaf, path->slots[0]);
- buf = kmalloc(item_size, GFP_NOFS);
+ buf = kvmalloc(item_size, GFP_NOFS);
if (!buf)
return -ENOMEM;
@@ -3981,7 +3981,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans,
btrfs_mark_buffer_dirty(trans, leaf);
BUG_ON(btrfs_leaf_free_space(leaf) < 0);
- kfree(buf);
+ kvfree(buf);
return 0;
}
diff --git a/fs/btrfs/delalloc-space.c b/fs/btrfs/delalloc-space.c
index d357ed7efd99..77781852e417 100644
--- a/fs/btrfs/delalloc-space.c
+++ b/fs/btrfs/delalloc-space.c
@@ -132,9 +132,7 @@ int btrfs_alloc_data_chunk_ondemand(const struct btrfs_inode *inode, u64 bytes)
/* Make sure bytes are sectorsize aligned */
bytes = ALIGN(bytes, fs_info->sectorsize);
- if (btrfs_is_free_space_inode(inode))
- flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE;
- else if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root))
+ if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root))
flush = BTRFS_RESERVE_FLUSH_ZONED_RELOCATION;
return btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), bytes, flush);
@@ -155,8 +153,6 @@ int btrfs_check_data_free_space(struct btrfs_inode *inode,
if (noflush)
flush = BTRFS_RESERVE_NO_FLUSH;
- else if (btrfs_is_free_space_inode(inode))
- flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE;
ret = btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), len, flush);
if (ret < 0)
@@ -326,15 +322,10 @@ int btrfs_delalloc_reserve_metadata(struct btrfs_inode *inode, u64 num_bytes,
int ret = 0;
/*
- * If we are a free space inode we need to not flush since we will be in
- * the middle of a transaction commit. We also don't need the delalloc
- * mutex since we won't race with anybody. We need this mostly to make
- * lockdep shut its filthy mouth.
- *
* If we have a transaction open (can happen if we call truncate_block
* from truncate), then we need FLUSH_LIMIT so we don't deadlock.
*/
- if (noflush || btrfs_is_free_space_inode(inode)) {
+ if (noflush) {
flush = BTRFS_RESERVE_NO_FLUSH;
} else {
if (current->journal_info)
diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c
index db2ffab0941a..1f20148c1fd4 100644
--- a/fs/btrfs/delayed-inode.c
+++ b/fs/btrfs/delayed-inode.c
@@ -6,6 +6,7 @@
#include <linux/slab.h>
#include <linux/iversion.h>
+#include <linux/error-injection.h>
#include "ctree.h"
#include "fs.h"
#include "messages.h"
@@ -686,7 +687,7 @@ static int btrfs_insert_delayed_item(struct btrfs_trans_handle *trans,
/*
* For delayed items to insert, we track reserved metadata bytes based
* on the number of leaves that we will use.
- * See btrfs_insert_delayed_dir_index() and
+ * See btrfs_insert_delayed_dir_index_prealloc() and
* btrfs_delayed_item_reserve_metadata()).
*/
ASSERT(first_item->bytes_reserved == 0);
@@ -1469,35 +1470,93 @@ static void btrfs_release_dir_index_item_space(struct btrfs_trans_handle *trans)
trans->bytes_reserved -= bytes;
}
-/* Will return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */
-int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans,
- const char *name, int name_len,
- struct btrfs_inode *dir,
- const struct btrfs_disk_key *disk_key, u8 flags,
- u64 index)
+/*
+ * Pre-allocate a delayed node and delayed item for a dir index insertion and
+ * copy the name into the item. Call this before modifying the btree so that
+ * ENOMEM can be returned before any on-disk state has changed.
+ *
+ * The returned prealloc is consumed by either
+ * btrfs_insert_delayed_dir_index_prealloc() or
+ * btrfs_free_delayed_dir_index_prealloc(); it must not be used afterwards.
+ *
+ * Returns a prealloc on success, ERR_PTR on allocation failure.
+ */
+struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir,
+ const char *name,
+ int name_len)
+{
+ struct btrfs_dir_index_prealloc *prealloc;
+ struct btrfs_delayed_node *node;
+ struct btrfs_delayed_item *item;
+
+ prealloc = kzalloc_obj(*prealloc, GFP_NOFS);
+ if (!prealloc)
+ return ERR_PTR(-ENOMEM);
+
+ node = btrfs_get_or_create_delayed_node(dir, &prealloc->tracker);
+ if (IS_ERR(node)) {
+ kfree(prealloc);
+ return ERR_CAST(node);
+ }
+
+ item = btrfs_alloc_delayed_item(sizeof(struct btrfs_dir_item) + name_len,
+ node, BTRFS_DELAYED_INSERTION_ITEM);
+ if (!item) {
+ btrfs_release_delayed_node(node, &prealloc->tracker);
+ kfree(prealloc);
+ return ERR_PTR(-ENOMEM);
+ }
+
+ memcpy(item->data + sizeof(struct btrfs_dir_item), name, name_len);
+
+ prealloc->node = node;
+ prealloc->item = item;
+ return prealloc;
+}
+ALLOW_ERROR_INJECTION(btrfs_prealloc_delayed_dir_index, ERRNO);
+
+/*
+ * Free resources from btrfs_prealloc_delayed_dir_index() when the btree
+ * insertion failed and we will not commit the delayed dir index. Does nothing
+ * if @prealloc is NULL.
+ */
+void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans,
+ struct btrfs_dir_index_prealloc *prealloc)
{
+ if (!prealloc)
+ return;
+
+ btrfs_release_delayed_item(prealloc->item);
+ btrfs_release_dir_index_item_space(trans);
+ btrfs_release_delayed_node(prealloc->node, &prealloc->tracker);
+ kfree(prealloc);
+}
+
+/*
+ * Commit a pre-allocated delayed dir index item. @prealloc must have been
+ * returned by btrfs_prealloc_delayed_dir_index(). This populates the item,
+ * adds it to the delayed node's rb-tree, and reserves metadata space. It
+ * cannot fail with ENOMEM. @prealloc is freed here in all cases.
+ *
+ * Return 0 or -EEXIST (index number collision, unexpected).
+ */
+int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans,
+ struct btrfs_inode *dir,
+ struct btrfs_dir_index_prealloc *prealloc,
+ const struct btrfs_disk_key *disk_key,
+ u8 flags, u64 index)
+{
+ struct btrfs_delayed_node *delayed_node = prealloc->node;
+ struct btrfs_ref_tracker *tracker = &prealloc->tracker;
+ struct btrfs_delayed_item *delayed_item = prealloc->item;
struct btrfs_fs_info *fs_info = trans->fs_info;
const unsigned int leaf_data_size = BTRFS_LEAF_DATA_SIZE(fs_info);
- struct btrfs_delayed_node *delayed_node;
- struct btrfs_ref_tracker delayed_node_tracker;
- struct btrfs_delayed_item *delayed_item;
+ const int name_len = delayed_item->data_len - sizeof(struct btrfs_dir_item);
struct btrfs_dir_item *dir_item;
bool reserve_leaf_space;
u32 data_len;
int ret;
- delayed_node = btrfs_get_or_create_delayed_node(dir, &delayed_node_tracker);
- if (IS_ERR(delayed_node))
- return PTR_ERR(delayed_node);
-
- delayed_item = btrfs_alloc_delayed_item(sizeof(*dir_item) + name_len,
- delayed_node,
- BTRFS_DELAYED_INSERTION_ITEM);
- if (!delayed_item) {
- ret = -ENOMEM;
- goto release_node;
- }
-
delayed_item->index = index;
dir_item = (struct btrfs_dir_item *)delayed_item->data;
@@ -1506,7 +1565,7 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans,
btrfs_set_stack_dir_data_len(dir_item, 0);
btrfs_set_stack_dir_name_len(dir_item, name_len);
btrfs_set_stack_dir_flags(dir_item, flags);
- memcpy((char *)(dir_item + 1), name, name_len);
+ /* Name was already copied by btrfs_prealloc_delayed_dir_index(). */
data_len = delayed_item->data_len + sizeof(struct btrfs_item);
@@ -1524,7 +1583,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans,
if (unlikely(ret)) {
btrfs_err(trans->fs_info,
"error adding delayed dir index item, name: %.*s, index: %llu, root: %llu, dir: %llu, dir->index_cnt: %llu, delayed_node->index_cnt: %llu, error: %pe",
- name_len, name, index, btrfs_root_id(delayed_node->root),
+ name_len,
+ (const char *)(dir_item + 1),
+ index, btrfs_root_id(delayed_node->root),
delayed_node->inode_id, dir->index_cnt,
delayed_node->index_cnt, ERR_PTR(ret));
btrfs_release_delayed_item(delayed_item);
@@ -1562,7 +1623,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans,
mutex_unlock(&delayed_node->mutex);
release_node:
- btrfs_release_delayed_node(delayed_node, &delayed_node_tracker);
+ /* Must release the node before freeing @tracker's containing struct. */
+ btrfs_release_delayed_node(delayed_node, tracker);
+ kfree(prealloc);
return ret;
}
@@ -1581,7 +1644,7 @@ static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node,
/*
* For delayed items to insert, we track reserved metadata bytes based
* on the number of leaves that we will use.
- * See btrfs_insert_delayed_dir_index() and
+ * See btrfs_insert_delayed_dir_index_prealloc() and
* btrfs_delayed_item_reserve_metadata()).
*/
ASSERT(item->bytes_reserved == 0);
diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h
index fc752863f89b..57ba96cfaf9c 100644
--- a/fs/btrfs/delayed-inode.h
+++ b/fs/btrfs/delayed-inode.h
@@ -115,11 +115,23 @@ struct btrfs_delayed_item {
};
void btrfs_init_delayed_root(struct btrfs_delayed_root *delayed_root);
-int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans,
- const char *name, int name_len,
- struct btrfs_inode *dir,
- const struct btrfs_disk_key *disk_key, u8 flags,
- u64 index);
+
+struct btrfs_dir_index_prealloc {
+ struct btrfs_delayed_node *node;
+ struct btrfs_ref_tracker tracker;
+ struct btrfs_delayed_item *item;
+};
+
+struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir,
+ const char *name,
+ int name_len);
+void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans,
+ struct btrfs_dir_index_prealloc *prealloc);
+int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans,
+ struct btrfs_inode *dir,
+ struct btrfs_dir_index_prealloc *prealloc,
+ const struct btrfs_disk_key *disk_key,
+ u8 flags, u64 index);
int btrfs_delete_delayed_dir_index(struct btrfs_trans_handle *trans,
struct btrfs_inode *dir, u64 index);
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
index 0284be0e4e82..cdfe093e5c5c 100644
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -235,7 +235,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
struct btrfs_device **device_out)
{
struct btrfs_fs_devices *fs_devices = fs_info->fs_devices;
- struct btrfs_device *device;
+ struct btrfs_device *device = NULL;
+ struct btrfs_device *tmp_device;
struct file *bdev_file;
struct block_device *bdev;
u64 devid = BTRFS_DEV_REPLACE_DEVID;
@@ -264,8 +265,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
sync_blockdev(bdev);
- list_for_each_entry(device, &fs_devices->devices, dev_list) {
- if (device->bdev == bdev) {
+ list_for_each_entry(tmp_device, &fs_devices->devices, dev_list) {
+ if (tmp_device->bdev == bdev) {
btrfs_err(fs_info,
"target device is in the filesystem!");
ret = -EEXIST;
@@ -285,6 +286,7 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
device = btrfs_alloc_device(NULL, &devid, NULL, device_path);
if (IS_ERR(device)) {
ret = PTR_ERR(device);
+ device = NULL;
goto error;
}
@@ -328,6 +330,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
error:
/* Undo the open-time freeze deny. */
+ if (device)
+ btrfs_free_device(device);
btrfs_release_device_allow_freeze(bdev_file);
return ret;
}
diff --git a/fs/btrfs/dir-item.c b/fs/btrfs/dir-item.c
index 84f1c64423d3..3a90736915af 100644
--- a/fs/btrfs/dir-item.c
+++ b/fs/btrfs/dir-item.c
@@ -106,8 +106,11 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans,
* Will return 0 or -ENOMEM
*/
int btrfs_insert_dir_item(struct btrfs_trans_handle *trans,
- const struct fscrypt_str *name, struct btrfs_inode *dir,
- const struct btrfs_key *location, u8 type, u64 index)
+ const struct fscrypt_str *name,
+ struct btrfs_inode *dir,
+ const struct btrfs_key *location, u8 type,
+ u64 index,
+ struct btrfs_dir_index_prealloc *prealloc)
{
int ret = 0;
int ret2 = 0;
@@ -119,17 +122,27 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans,
struct btrfs_key key;
struct btrfs_disk_key disk_key;
u32 data_size;
+ const bool need_delayed_index = (root != root->fs_info->tree_root);
key.objectid = btrfs_ino(dir);
key.type = BTRFS_DIR_ITEM_KEY;
key.offset = btrfs_name_hash(name->name, name->len);
path = btrfs_alloc_path();
- if (!path)
- return -ENOMEM;
+ if (!path) {
+ ret = -ENOMEM;
+ goto out_free_prealloc;
+ }
btrfs_cpu_key_to_disk(&disk_key, location);
+ /* Pre-allocate the delayed dir index before modifying the btree. */
+ if (need_delayed_index && !prealloc) {
+ prealloc = btrfs_prealloc_delayed_dir_index(dir, name->name, name->len);
+ if (IS_ERR(prealloc))
+ return PTR_ERR(prealloc);
+ }
+
data_size = sizeof(*dir_item) + name->len;
dir_item = insert_with_overflow(trans, root, path, &key, data_size,
name->name, name->len);
@@ -137,7 +150,7 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans,
ret = PTR_ERR(dir_item);
if (ret == -EEXIST)
goto second_insert;
- goto out_free;
+ goto out_free_prealloc;
}
if (IS_ENCRYPTED(&dir->vfs_inode))
@@ -154,21 +167,21 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans,
write_extent_buffer(leaf, name->name, name_ptr, name->len);
second_insert:
- /* FIXME, use some real flag for selecting the extra index */
- if (root == root->fs_info->tree_root) {
+ if (!need_delayed_index) {
ret = 0;
- goto out_free;
+ goto out_free_prealloc;
}
btrfs_release_path(path);
- ret2 = btrfs_insert_delayed_dir_index(trans, name->name, name->len, dir,
- &disk_key, type, index);
-out_free:
+ ret2 = btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc,
+ &disk_key, type, index);
if (ret)
return ret;
- if (ret2)
- return ret2;
- return 0;
+ return ret2;
+
+out_free_prealloc:
+ btrfs_free_delayed_dir_index_prealloc(trans, prealloc);
+ return ret;
}
static struct btrfs_dir_item *btrfs_lookup_match_dir(
diff --git a/fs/btrfs/dir-item.h b/fs/btrfs/dir-item.h
index e52174a8baf9..8d22976a0a77 100644
--- a/fs/btrfs/dir-item.h
+++ b/fs/btrfs/dir-item.h
@@ -13,12 +13,14 @@ struct btrfs_path;
struct btrfs_inode;
struct btrfs_root;
struct btrfs_trans_handle;
+struct btrfs_dir_index_prealloc;
int btrfs_check_dir_item_collision(struct btrfs_root *root, u64 dir_ino,
const struct fscrypt_str *name);
int btrfs_insert_dir_item(struct btrfs_trans_handle *trans,
const struct fscrypt_str *name, struct btrfs_inode *dir,
- const struct btrfs_key *location, u8 type, u64 index);
+ const struct btrfs_key *location, u8 type, u64 index,
+ struct btrfs_dir_index_prealloc *prealloc);
struct btrfs_dir_item *btrfs_lookup_dir_item(struct btrfs_trans_handle *trans,
struct btrfs_root *root,
struct btrfs_path *path, u64 dir,
@@ -53,5 +55,4 @@ static inline u64 btrfs_name_hash(const char *name, int len)
{
return crc32c((u32)~1, name, len);
}
-
#endif
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index dc7ad92876c0..7fd8babd74a6 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -125,15 +125,13 @@ int btrfs_buffer_uptodate(struct extent_buffer *eb, u64 parent_transid,
return 1;
}
- if (btrfs_header_generation(eb) != parent_transid) {
- btrfs_err_rl(eb->fs_info,
+ btrfs_err_rl(eb->fs_info,
"parent transid verify failed on logical %llu mirror %u wanted %llu found %llu",
- eb->start, eb->read_mirror,
- parent_transid, btrfs_header_generation(eb));
- clear_extent_buffer_uptodate(eb);
- return 0;
- }
- return 1;
+ eb->start, eb->read_mirror,
+ parent_transid, btrfs_header_generation(eb));
+ clear_extent_buffer_uptodate(eb);
+
+ return 0;
}
static bool btrfs_supported_super_csum(u16 csum_type)
@@ -176,19 +174,24 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb,
int mirror_num)
{
struct btrfs_fs_info *fs_info = eb->fs_info;
- const u32 step = min(fs_info->nodesize, PAGE_SIZE);
- const u32 nr_steps = eb->len / step;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
+ struct btrfs_bio *bbio;
+ int ret;
if (sb_rdonly(fs_info->sb))
return -EROFS;
+ /*
+ * This bbio is only to queue all pages for btrfs_repair_bbio_failure().
+ * Thus it will never get its endio called.
+ */
+ bbio = btrfs_bio_alloc(max(1, fs_info->nodesize >> PAGE_SHIFT), REQ_OP_READ,
+ BTRFS_I(fs_info->btree_inode), eb->start, NULL, NULL);
+ bbio->bio.bi_iter.bi_sector = eb->start >> SECTOR_SHIFT;
for (int i = 0; i < num_extent_pages(eb); i++) {
struct folio *folio = eb->folios[i];
/* No large folio support yet. */
ASSERT(folio_order(folio) == 0);
- ASSERT(i < nr_steps);
/*
* For nodesize < page size, there is just one paddr, with some
@@ -197,11 +200,17 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb,
* For nodesize >= page size, it's one or more paddrs, and eb->start
* must be aligned to page boundary.
*/
- paddrs[i] = page_to_phys(&folio->page) + offset_in_page(eb->start);
+ ret = bio_add_page(&bbio->bio, &folio->page, min(PAGE_SIZE, fs_info->nodesize),
+ offset_in_page(eb->start));
+ ASSERT(ret == min(PAGE_SIZE, fs_info->nodesize));
}
+ /* Since the bbio is never submitted, we have to save the iter manually. */
+ bbio->saved_iter = bbio->bio.bi_iter;
- return btrfs_repair_io_failure(fs_info, 0, eb->start, eb->len,
- eb->start, paddrs, step, mirror_num);
+ ret = btrfs_repair_bbio_failure(bbio, &bbio->saved_iter, fs_info->nodesize,
+ mirror_num);
+ bio_put(&bbio->bio);
+ return ret;
}
/*
@@ -1485,7 +1494,9 @@ static int cleaner_kthread(void *arg)
btrfs_run_delayed_iputs(fs_info);
+ set_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags);
again = btrfs_clean_one_deleted_snapshot(fs_info);
+ clear_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags);
mutex_unlock(&fs_info->cleaner_mutex);
/*
@@ -1769,7 +1780,6 @@ static void btrfs_stop_all_workers(struct btrfs_fs_info *fs_info)
if (fs_info->rmw_workers)
destroy_workqueue(fs_info->rmw_workers);
btrfs_destroy_workqueue(fs_info->endio_write_workers);
- btrfs_destroy_workqueue(fs_info->endio_freespace_worker);
btrfs_destroy_workqueue(fs_info->delayed_workers);
btrfs_destroy_workqueue(fs_info->caching_workers);
btrfs_destroy_workqueue(fs_info->flush_workers);
@@ -1980,9 +1990,6 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info)
fs_info->endio_write_workers =
btrfs_alloc_workqueue(fs_info, "endio-write", flags,
max_active, 2);
- fs_info->endio_freespace_worker =
- btrfs_alloc_workqueue(fs_info, "freespace-write", flags,
- max_active, 0);
fs_info->delayed_workers =
btrfs_alloc_workqueue(fs_info, "delayed-meta", flags,
max_active, 0);
@@ -1995,8 +2002,7 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info)
if (!(fs_info->workers &&
fs_info->delalloc_workers && fs_info->flush_workers &&
fs_info->endio_workers && fs_info->endio_meta_workers &&
- fs_info->endio_write_workers &&
- fs_info->endio_freespace_worker && fs_info->rmw_workers &&
+ fs_info->endio_write_workers && fs_info->rmw_workers &&
fs_info->caching_workers && fs_info->fixup_workers &&
fs_info->delayed_workers && fs_info->qgroup_rescan_workers &&
fs_info->discard_ctl.discard_workers)) {
@@ -2374,7 +2380,7 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info,
}
ret = btrfs_check_chunk_valid(fs_info, NULL, chunk, key.offset,
sectorsize);
- if (ret < 0)
+ if (unlikely(ret < 0))
return ret;
cur += btrfs_chunk_item_size(num_stripes);
}
@@ -3063,7 +3069,6 @@ static int btrfs_cleanup_fs_roots(struct btrfs_fs_info *fs_info)
int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info)
{
int ret;
- const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE);
bool rebuild_free_space_tree = false;
if (btrfs_test_opt(fs_info, CLEAR_CACHE) &&
@@ -3158,8 +3163,8 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info)
}
}
- if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) {
- ret = btrfs_set_free_space_cache_v1_active(fs_info, cache_opt);
+ if (btrfs_free_space_cache_v1_active(fs_info)) {
+ ret = btrfs_cleanup_free_space_cache_v1(fs_info);
if (ret)
return ret;
}
@@ -3275,20 +3280,6 @@ int btrfs_check_features(struct btrfs_fs_info *fs_info, bool is_rw_mount)
return -EINVAL;
}
- /*
- * Subpage/bs > ps runtime limitation on v1 cache.
- *
- * V1 space cache still has some hard coded PAGE_SIZE usage, while
- * we're already defaulting to v2 cache, no need to bother v1 as it's
- * going to be deprecated anyway.
- */
- if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) {
- btrfs_warn(fs_info,
- "v1 space cache is not supported for page size %lu with sectorsize %u",
- PAGE_SIZE, fs_info->sectorsize);
- return -EINVAL;
- }
-
/* This can be called by remount, we need to protect the super block. */
spin_lock(&fs_info->super_lock);
btrfs_set_super_incompat_flags(disk_super, incompat);
@@ -4217,8 +4208,6 @@ int write_all_supers(struct btrfs_trans_handle *trans)
total_errors++;
}
if (unlikely(total_errors > max_errors)) {
- btrfs_err(fs_info, "%d errors while writing supers",
- total_errors);
mutex_unlock(&fs_info->fs_devices->device_list_mutex);
/* FUA is masked off if unsupported and can't be the reason */
@@ -4446,9 +4435,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
* to finish an ordered extent - end_bbio_compressed_write()
* calls btrfs_finish_ordered_extent() which in turns does a call to
* btrfs_queue_ordered_fn(), and that queues the ordered extent
- * completion either in the endio_write_workers work queue or in the
- * fs_info->endio_freespace_worker work queue. We flush those queues
- * below, so before we flush them we must flush this queue for the
+ * completion in the endio_write_workers work queue. We flush that
+ * queue below, so before we flush it we must flush this queue for the
* workers of compressed writes.
*/
flush_workqueue(fs_info->endio_workers);
@@ -4474,8 +4462,6 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
* btrfs_finish_ordered_io() when we are unmounting).
*/
btrfs_flush_workqueue(fs_info->endio_write_workers);
- /* Ordered extents for free space inodes. */
- btrfs_flush_workqueue(fs_info->endio_freespace_worker);
/*
* Run delayed iputs in case an async reclaim worker is waiting for them
* to be run as mentioned above.
@@ -4864,26 +4850,6 @@ static void btrfs_destroy_pinned_extent(struct btrfs_fs_info *fs_info,
}
}
-static void btrfs_cleanup_bg_io(struct btrfs_block_group *cache)
-{
- struct inode *inode;
-
- inode = cache->io_ctl.inode;
- if (inode) {
- unsigned int nofs_flag;
-
- nofs_flag = memalloc_nofs_save();
- invalidate_inode_pages2(inode->i_mapping);
- memalloc_nofs_restore(nofs_flag);
-
- BTRFS_I(inode)->generation = 0;
- cache->io_ctl.inode = NULL;
- iput(inode);
- }
- ASSERT(cache->io_ctl.pages == NULL);
- btrfs_put_block_group(cache);
-}
-
void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans,
struct btrfs_fs_info *fs_info)
{
@@ -4895,40 +4861,13 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans,
struct btrfs_block_group,
dirty_list);
- if (!list_empty(&cache->io_list)) {
- spin_unlock(&cur_trans->dirty_bgs_lock);
- list_del_init(&cache->io_list);
- btrfs_cleanup_bg_io(cache);
- spin_lock(&cur_trans->dirty_bgs_lock);
- }
-
list_del_init(&cache->dirty_list);
- spin_lock(&cache->lock);
- cache->disk_cache_state = BTRFS_DC_ERROR;
- spin_unlock(&cache->lock);
-
spin_unlock(&cur_trans->dirty_bgs_lock);
btrfs_put_block_group(cache);
btrfs_dec_delayed_refs_rsv_bg_updates(fs_info);
spin_lock(&cur_trans->dirty_bgs_lock);
}
spin_unlock(&cur_trans->dirty_bgs_lock);
-
- /*
- * Refer to the definition of io_bgs member for details why it's safe
- * to use it without any locking
- */
- while (!list_empty(&cur_trans->io_bgs)) {
- cache = list_first_entry(&cur_trans->io_bgs,
- struct btrfs_block_group,
- io_list);
-
- list_del_init(&cache->io_list);
- spin_lock(&cache->lock);
- cache->disk_cache_state = BTRFS_DC_ERROR;
- spin_unlock(&cache->lock);
- btrfs_cleanup_bg_io(cache);
- }
}
static void btrfs_free_all_qgroup_pertrans(struct btrfs_fs_info *fs_info)
@@ -4964,7 +4903,6 @@ void btrfs_cleanup_one_transaction(struct btrfs_transaction *cur_trans)
btrfs_cleanup_dirty_bgs(cur_trans, fs_info);
ASSERT(list_empty(&cur_trans->dirty_bgs));
- ASSERT(list_empty(&cur_trans->io_bgs));
list_for_each_entry_safe(dev, tmp, &cur_trans->dev_update_list,
post_commit_list) {
diff --git a/fs/btrfs/extent-io-tree.c b/fs/btrfs/extent-io-tree.c
index d6df11f6088c..992b8b42bdb4 100644
--- a/fs/btrfs/extent-io-tree.c
+++ b/fs/btrfs/extent-io-tree.c
@@ -751,7 +751,7 @@ hit_next:
btrfs_split_delalloc_extent(tree->inode, state, start);
/*
- * Temporarilly ajdust this state's range to match the
+ * Temporarily ajdust this state's range to match the
* range for which we are clearing bits.
*/
state->start = start;
diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c
index d6a4390ee34a..a0d5ab03aae2 100644
--- a/fs/btrfs/extent-tree.c
+++ b/fs/btrfs/extent-tree.c
@@ -6315,6 +6315,16 @@ int btrfs_drop_snapshot(struct btrfs_root *root, bool update_ref, bool for_reloc
set_bit(BTRFS_ROOT_DELETING, &root->state);
unfinished_drop = test_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state);
+ /*
+ * For subvolume dropping, check if the subvolume is large enough so
+ * that we need to mark qgroup inconsistent to avoid long qgroup stall.
+ *
+ * Even for a subvolume without any snapshot, there can still be
+ * a lot of qgroup records queued into one transaction.
+ */
+ if (!for_reloc)
+ btrfs_qgroup_check_tree_drop(fs_info, rootid,
+ btrfs_header_level(root->node));
if (btrfs_disk_key_objectid(&root_item->drop_progress) == 0) {
level = btrfs_header_level(root->node);
path->nodes[level] = btrfs_lock_root_node(root);
diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c
index d7600e5fa3d9..55e9144d4759 100644
--- a/fs/btrfs/extent_io.c
+++ b/fs/btrfs/extent_io.c
@@ -110,14 +110,14 @@ struct btrfs_bio_ctrl {
* make the decision when submitting the bio.
*
* The pattern between do_readpage(), submit_one_bio() and
- * submit_extent_folio() is quite subtle, so tracking this is tricky.
+ * submit_one_block() is quite subtle, so tracking this is tricky.
*
* As we process extent E, we might submit a bio with existing built up
* extents before adding E to a new bio, or we might just add E to the
* bio. As a result, E's generation could apply to the current bio or
* to the next one, so we need to be careful to update the bio_ctrl's
* generation with E's only when we are sure E is added to bio_ctrl->bbio
- * in submit_extent_folio().
+ * in submit_one_block().
*
* See the comment in btrfs_lookup_bio_sums() for more detail on the
* need for this optimization.
@@ -797,118 +797,95 @@ static int alloc_new_bio(struct btrfs_inode *inode,
}
/*
- * @disk_bytenr: logical bytenr where the write will be
- * @page: page to add to the bio
- * @size: portion of page that we want to write to
- * @pg_offset: offset of the new bio or to check whether we are adding
- * a contiguous page to the previous one
+ * @disk_bytenr: logical bytenr where the read/write will be
+ * @folio: the folio the block belongs to
+ * @pg_offset: the offset inside the folio
* @read_em_generation: generation of the extent_map we are submitting
* (only used for read)
*
- * The will either add the page into the existing @bio_ctrl->bbio, or allocate a
+ * This will either add the block into the existing @bio_ctrl->bbio, or allocate a
* new one in @bio_ctrl->bbio.
* The mirror number for this IO should already be initialized in
* @bio_ctrl->mirror_num.
*
- * Return the number of bytes that are queued into a bio.
- * If the returned bytes is smaller than @size, it means we hit a critical error
- * for data write, where there is no ordered extent for the range.
+ * Return 0 if the block is queued or submitted.
+ * Return <0 for error.
*/
-static unsigned int submit_extent_folio(struct btrfs_bio_ctrl *bio_ctrl,
- u64 disk_bytenr, struct folio *folio,
- size_t size, unsigned long pg_offset,
- u64 read_em_generation)
+static int submit_one_block(struct btrfs_bio_ctrl *bio_ctrl,
+ u64 disk_bytenr, struct folio *folio,
+ unsigned long pg_offset, u64 read_em_generation)
{
struct btrfs_inode *inode = folio_to_inode(folio);
+ const struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ const u32 blocksize = fs_info->sectorsize;
loff_t file_offset = folio_pos(folio) + pg_offset;
- unsigned int queued = 0;
- ASSERT(pg_offset + size <= folio_size(folio));
+ ASSERT(pg_offset + blocksize <= folio_size(folio));
ASSERT(bio_ctrl->end_io_func);
if (bio_ctrl->bbio &&
!btrfs_bio_is_contig(bio_ctrl, disk_bytenr, file_offset))
submit_one_bio(bio_ctrl);
- do {
- u32 len = size;
-
- /* Allocate new bio if needed */
- if (!bio_ctrl->bbio) {
- int ret;
-
- ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset);
- if (ret < 0)
- break;
- }
-
- /* Cap to the current ordered extent boundary if there is one. */
- if (len > bio_ctrl->len_to_oe_boundary) {
- ASSERT(bio_ctrl->compress_type == BTRFS_COMPRESS_NONE);
- ASSERT(is_data_inode(inode));
- len = bio_ctrl->len_to_oe_boundary;
- }
+again:
+ /* Allocate new bio if needed */
+ if (!bio_ctrl->bbio) {
+ int ret;
- if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, len, pg_offset)) {
- /* bio full: move on to a new one */
- submit_one_bio(bio_ctrl);
- continue;
- }
- /*
- * Now that the folio is definitely added to the bio, include its
- * generation in the max generation calculation.
- */
- bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation);
- bio_ctrl->next_file_offset += len;
+ ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset);
+ if (ret < 0)
+ return ret;
+ }
- if (bio_ctrl->wbc)
- wbc_account_cgroup_owner(bio_ctrl->wbc, folio, len);
+ if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, blocksize, pg_offset)) {
+ /* bio full: move on to a new one */
+ submit_one_bio(bio_ctrl);
+ goto again;
+ }
- size -= len;
- pg_offset += len;
- disk_bytenr += len;
- file_offset += len;
- queued += len;
+ /*
+ * Now that the folio is definitely added to the bio, include its
+ * generation in the max generation calculation.
+ */
+ bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation);
+ bio_ctrl->next_file_offset += blocksize;
- /*
- * len_to_oe_boundary defaults to U32_MAX, which isn't folio or
- * sector aligned. alloc_new_bio() then sets it to the end of
- * our ordered extent for writes into zoned devices.
- *
- * When len_to_oe_boundary is tracking an ordered extent, we
- * trust the ordered extent code to align things properly, and
- * the check above to cap our write to the ordered extent
- * boundary is correct.
- *
- * When len_to_oe_boundary is U32_MAX, the cap above would
- * result in a 4095 byte IO for the last folio right before
- * we hit the bio limit of UINT_MAX. bio_add_folio() has all
- * the checks required to make sure we don't overflow the bio,
- * and we should just ignore len_to_oe_boundary completely
- * unless we're using it to track an ordered extent.
- *
- * It's pretty hard to make a bio sized U32_MAX, but it can
- * happen when the page cache is able to feed us contiguous
- * folios for large extents.
- */
- if (bio_ctrl->len_to_oe_boundary != U32_MAX)
- bio_ctrl->len_to_oe_boundary -= len;
+ if (bio_ctrl->wbc)
+ wbc_account_cgroup_owner(bio_ctrl->wbc, folio, blocksize);
- /* Ordered extent boundary: move on to a new bio. */
- if (bio_ctrl->len_to_oe_boundary == 0)
- submit_one_bio(bio_ctrl);
- /*
- * If we have accumulated decent amount of IO, send it to the
- * block layer so that IO can run while we are accumulating
- * more folios to write.
- */
- else if (bio_ctrl->wbc &&
- bio_ctrl->bbio->bio.bi_iter.bi_size >=
- inode->root->fs_info->writeback_bio_size)
- submit_one_bio(bio_ctrl);
+ /*
+ * len_to_oe_boundary defaults to U32_MAX, which isn't folio or sector
+ * aligned. alloc_new_bio() then sets it to the end of our ordered
+ * extent for writes into zoned devices.
+ *
+ * When len_to_oe_boundary is tracking an ordered extent, the
+ * len_to_oe_boundary should follow that OE and never go beyond the max
+ * extent size (128MiB).
+ *
+ * When len_to_oe_boundary is U32_MAX, decreasing the length by
+ * blocksize will never make it reach 0, thus skipping the later
+ * submit_one_bio() call. So if len_to_oe_boundary() is not tracking
+ * an OE, do not decrease it.
+ *
+ * It's pretty hard to make a bio sized U32_MAX, but it can happen when
+ * the page cache is able to feed us contiguous folios for large
+ * extents.
+ */
+ if (bio_ctrl->len_to_oe_boundary != U32_MAX)
+ bio_ctrl->len_to_oe_boundary -= blocksize;
- } while (size);
- return queued;
+ /* Ordered extent boundary: move on to a new bio. */
+ if (bio_ctrl->len_to_oe_boundary == 0)
+ submit_one_bio(bio_ctrl);
+ /*
+ * If we have accumulated decent amount of IO, send it to the block
+ * layer so that IO can run while we are accumulating more folios to
+ * write.
+ */
+ else if (bio_ctrl->wbc &&
+ bio_ctrl->bbio->bio.bi_iter.bi_size >= fs_info->writeback_bio_size)
+ submit_one_bio(bio_ctrl);
+ return 0;
}
static int attach_extent_buffer_folio(struct extent_buffer *eb,
@@ -1092,7 +1069,6 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached,
u64 disk_bytenr;
u64 block_start;
u64 em_gen;
- unsigned int queued;
ASSERT(IS_ALIGNED(cur, fs_info->sectorsize));
if (cur >= last_byte) {
@@ -1206,10 +1182,9 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached,
if (force_bio_submit)
submit_one_bio(bio_ctrl);
- queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, blocksize,
- pg_offset, em_gen);
+ ret = submit_one_block(bio_ctrl, disk_bytenr, folio, pg_offset, em_gen);
/* Read submission should not fail. */
- ASSERT(queued == blocksize);
+ ASSERT(ret == 0);
}
return 0;
}
@@ -1808,33 +1783,51 @@ out:
return 0;
}
+static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbio,
+ u64 filepos)
+{
+ struct btrfs_ordered_extent *oe;
+
+ if (!bbio || !bbio->ordered)
+ return NULL;
+
+ oe = bbio->ordered;
+ if (!in_range(filepos, oe->file_offset, oe->num_bytes))
+ return NULL;
+
+ refcount_inc(&oe->refs);
+ return oe;
+}
+
/*
* Return 0 if we have submitted or queued the sector for submission.
* Return <0 for critical errors, and the involved sector will be cleaned up.
*
* Caller should make sure filepos < i_size and handle filepos >= i_size case.
*/
-static int submit_one_sector(struct btrfs_inode *inode,
- struct folio *folio,
- u64 filepos, struct btrfs_bio_ctrl *bio_ctrl,
- loff_t i_size)
+static int submit_write_sector(struct btrfs_inode *inode,
+ struct folio *folio,
+ u64 filepos, struct btrfs_bio_ctrl *bio_ctrl,
+ loff_t i_size)
{
struct btrfs_fs_info *fs_info = inode->root->fs_info;
- struct extent_map *em;
+ struct btrfs_ordered_extent *oe;
u64 block_start;
u64 disk_bytenr;
u64 extent_offset;
- u64 em_end;
const u32 sectorsize = fs_info->sectorsize;
- unsigned int queued;
+ int ret;
ASSERT(IS_ALIGNED(filepos, sectorsize));
/* @filepos >= i_size case should be handled by the caller. */
ASSERT(filepos < i_size);
- em = btrfs_get_extent(inode, NULL, filepos, sectorsize);
- if (IS_ERR(em)) {
+ /* Try to reuse the existing OE from bbio first. */
+ oe = get_oe_from_bbio(bio_ctrl->bbio, filepos);
+ if (!oe)
+ oe = btrfs_lookup_ordered_extent(inode, filepos);
+ if (unlikely(!oe)) {
/*
* bio_ctrl may contain a bio crossing several folios.
* Submit it immediately so that the bio has a chance
@@ -1857,31 +1850,25 @@ static int submit_one_sector(struct btrfs_inode *inode,
*/
btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize,
false);
- return PTR_ERR(em);
+ btrfs_err_rl(fs_info,
+ "no ordered extent for root %lld ino %llu filepos %llu",
+ btrfs_root_id(inode->root), btrfs_ino(inode),
+ filepos);
+ return -EUCLEAN;
}
- extent_offset = filepos - em->start;
- em_end = btrfs_extent_map_end(em);
- ASSERT(filepos <= em_end);
- ASSERT(IS_ALIGNED(em->start, sectorsize));
- ASSERT(IS_ALIGNED(em->len, sectorsize));
+ extent_offset = filepos - oe->file_offset;
+ ASSERT(filepos < oe->file_offset + oe->num_bytes);
+ ASSERT(IS_ALIGNED(oe->file_offset, sectorsize));
+ ASSERT(IS_ALIGNED(oe->num_bytes, sectorsize));
+ ASSERT(oe->compress_type == BTRFS_COMPRESS_NONE);
+ ASSERT(!test_bit(BTRFS_ORDERED_COMPRESSED, &oe->flags));
- block_start = btrfs_extent_map_block_start(em);
- disk_bytenr = btrfs_extent_map_block_start(em) + extent_offset;
+ block_start = oe->disk_bytenr + oe->offset;
+ disk_bytenr = block_start + extent_offset;
- ASSERT(!btrfs_extent_map_is_compressed(em));
- ASSERT(block_start != EXTENT_MAP_HOLE);
- ASSERT(block_start != EXTENT_MAP_INLINE);
+ btrfs_put_ordered_extent(oe);
- btrfs_free_extent_map(em);
- em = NULL;
-
- /*
- * Although the PageDirty bit is cleared before entering this
- * function, subpage dirty bit is not cleared.
- * So clear subpage dirty bit here so next time we won't submit
- * a folio for a range already written to disk.
- */
btrfs_folio_clear_dirty(fs_info, folio, filepos, sectorsize);
btrfs_folio_set_writeback(fs_info, folio, filepos, sectorsize);
/*
@@ -1892,13 +1879,17 @@ static int submit_one_sector(struct btrfs_inode *inode,
*/
ASSERT(folio_test_writeback(folio));
- queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio,
- sectorsize, filepos - folio_pos(folio), 0);
- if (unlikely(queued < sectorsize)) {
+ ret = submit_one_block(bio_ctrl, disk_bytenr, folio,
+ offset_in_folio(folio, filepos), 0);
+ if (unlikely(ret < 0)) {
btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize);
btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize,
false);
- return -EUCLEAN;
+ btrfs_err_rl(fs_info,
+ "failed to queue sector for root %lld ino %llu filepos %llu: %pe",
+ btrfs_root_id(inode->root),
+ btrfs_ino(inode), filepos, ERR_PTR(ret));
+ return ret;
}
return 0;
}
@@ -1983,7 +1974,7 @@ static noinline_for_stack int extent_writepage_io(struct btrfs_inode *inode,
btrfs_folio_clear_dirty(fs_info, folio, cur, fs_info->sectorsize);
continue;
}
- ret = submit_one_sector(inode, folio, cur, bio_ctrl, i_size);
+ ret = submit_write_sector(inode, folio, cur, bio_ctrl, i_size);
if (unlikely(ret < 0)) {
if (!found_error)
found_error = ret;
diff --git a/fs/btrfs/extent_map.c b/fs/btrfs/extent_map.c
index 6ad7b39ae358..86d9c6f5ff4b 100644
--- a/fs/btrfs/extent_map.c
+++ b/fs/btrfs/extent_map.c
@@ -1220,6 +1220,14 @@ static struct btrfs_inode *find_first_inode_to_shrink(struct btrfs_root *root,
tree = &inode->extent_tree;
/*
+ * Most inodes have no extent maps, so check without the lock.
+ * The race is harmless: a false empty just defers the inode to
+ * a later scan, and a false non-empty is caught under the lock.
+ */
+ if (data_race(RB_EMPTY_ROOT(&tree->root)))
+ goto next;
+
+ /*
* We want to be fast so if the lock is busy we don't want to
* spend time waiting for it (some task is about to do IO for
* the inode).
diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c
index cf50fd623f41..ff8f8cad00fc 100644
--- a/fs/btrfs/file-item.c
+++ b/fs/btrfs/file-item.c
@@ -397,17 +397,6 @@ int btrfs_lookup_bio_sums(struct btrfs_bio *bbio)
path->reada = READA_FORWARD;
/*
- * the free space stuff is only read when it hasn't been
- * updated in the current transaction. So, we can safely
- * read from the commit root and sidestep a nasty deadlock
- * between reading the free space cache and updating the csum tree.
- */
- if (btrfs_is_free_space_inode(inode)) {
- path->search_commit_root = true;
- path->skip_locking = true;
- }
-
- /*
* If we are searching for a csum of an extent from a past
* transaction, we can search in the commit root and reduce
* lock contention on the csum tree extent buffers.
@@ -797,29 +786,20 @@ fail:
return ret;
}
-static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src)
+static void csum_one_bio(struct btrfs_bio *bbio)
{
struct btrfs_inode *inode = bbio->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
- struct bio *bio = &bbio->bio;
struct btrfs_ordered_sum *sums = bbio->sums;
- struct bvec_iter iter = *src;
- phys_addr_t paddr;
const u32 blocksize = fs_info->sectorsize;
- const u32 step = min(blocksize, PAGE_SIZE);
- const u32 nr_steps = blocksize / step;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
- u32 offset = 0;
int index = 0;
- btrfs_bio_for_each_block(paddr, bio, &iter, step) {
- paddrs[(offset / step) % nr_steps] = paddr;
- offset += step;
+ for (struct bvec_iter *iter = &bbio->csum_saved_iter;
+ iter->bi_size;
+ bio_advance_iter(&bbio->bio, iter, blocksize)) {
+ btrfs_csum_one_bio_block(fs_info, &bbio->bio, iter, sums->sums + index);
- if (IS_ALIGNED(offset, blocksize)) {
- btrfs_calculate_block_csum_pages(fs_info, paddrs, sums->sums + index);
- index += fs_info->csum_size;
- }
+ index += fs_info->csum_size;
}
}
@@ -828,9 +808,8 @@ static void csum_one_bio_work(struct work_struct *work)
struct btrfs_bio *bbio = container_of(work, struct btrfs_bio, csum_work);
ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE);
- ASSERT(bbio->async_csum == true);
- csum_one_bio(bbio, &bbio->csum_saved_iter);
- complete(&bbio->csum_done);
+ csum_one_bio(bbio);
+ bio_endio(&bbio->bio);
}
/*
@@ -859,15 +838,14 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async)
bbio->sums = sums;
btrfs_add_ordered_sum(ordered, sums);
+ bbio->csum_saved_iter = bio->bi_iter;
if (!async) {
- csum_one_bio(bbio, &bbio->bio.bi_iter);
+ csum_one_bio(bbio);
return 0;
}
- init_completion(&bbio->csum_done);
- bbio->async_csum = true;
- bbio->csum_saved_iter = bbio->bio.bi_iter;
+ bio_inc_remaining(bio);
INIT_WORK(&bbio->csum_work, csum_one_bio_work);
- schedule_work(&bbio->csum_work);
+ queue_work(fs_info->endio_workers, &bbio->csum_work);
return 0;
}
diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c
index e2af75a205ea..2a40167c3fb6 100644
--- a/fs/btrfs/free-space-cache.c
+++ b/fs/btrfs/free-space-cache.c
@@ -9,7 +9,6 @@
#include <linux/slab.h>
#include <linux/math64.h>
#include <linux/ratelimit.h>
-#include <linux/error-injection.h>
#include <linux/sched/mm.h>
#include <linux/string_choices.h>
#include "extent-tree.h"
@@ -23,7 +22,6 @@
#include "space-info.h"
#include "block-group.h"
#include "discard.h"
-#include "subpage.h"
#include "inode-item.h"
#include "accessors.h"
#include "file-item.h"
@@ -38,12 +36,6 @@
static struct kmem_cache *btrfs_free_space_cachep;
static struct kmem_cache *btrfs_free_space_bitmap_cachep;
-struct btrfs_trim_range {
- u64 start;
- u64 bytes;
- struct list_head list;
-};
-
static int link_free_space(struct btrfs_free_space_ctl *ctl,
struct btrfs_free_space *info);
static void unlink_free_space(struct btrfs_free_space_ctl *ctl,
@@ -57,11 +49,6 @@ static void bitmap_clear_bits(struct btrfs_free_space_ctl *ctl,
struct btrfs_free_space *info, u64 offset,
u64 bytes, bool update_stats);
-static void btrfs_crc32c_final(u32 crc, u8 *result)
-{
- put_unaligned_le32(~crc, result);
-}
-
static void __btrfs_remove_free_space_cache(struct btrfs_free_space_ctl *ctl)
{
struct btrfs_free_space *info;
@@ -123,10 +110,6 @@ static struct inode *__lookup_free_space_inode(struct btrfs_root *root,
if (IS_ERR(inode))
return ERR_CAST(inode);
- mapping_set_gfp_mask(inode->vfs_inode.i_mapping,
- mapping_gfp_constraint(inode->vfs_inode.i_mapping,
- ~(__GFP_FS | __GFP_HIGHMEM)));
-
return &inode->vfs_inode;
}
@@ -135,7 +118,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group,
{
struct btrfs_fs_info *fs_info = block_group->fs_info;
struct inode *inode = NULL;
- u32 flags = BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW;
spin_lock(&block_group->lock);
if (block_group->inode)
@@ -150,13 +132,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group,
return inode;
spin_lock(&block_group->lock);
- if (!((BTRFS_I(inode)->flags & flags) == flags)) {
- btrfs_info(fs_info, "Old style space inode found, converting.");
- BTRFS_I(inode)->flags |= BTRFS_INODE_NODATASUM |
- BTRFS_INODE_NODATACOW;
- block_group->disk_cache_state = BTRFS_DC_CLEAR;
- }
-
if (!test_and_set_bit(BLOCK_GROUP_FLAG_IREF, &block_group->runtime_flags))
block_group->inode = BTRFS_I(igrab(inode));
spin_unlock(&block_group->lock);
@@ -164,78 +139,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group,
return inode;
}
-static int __create_free_space_inode(struct btrfs_root *root,
- struct btrfs_trans_handle *trans,
- struct btrfs_path *path,
- u64 ino, u64 offset)
-{
- struct btrfs_key key;
- struct btrfs_disk_key disk_key;
- struct btrfs_free_space_header *header;
- struct btrfs_inode_item *inode_item;
- struct extent_buffer *leaf;
- /* We inline CRCs for the free disk space cache */
- const u64 flags = BTRFS_INODE_NOCOMPRESS | BTRFS_INODE_PREALLOC |
- BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW;
- int ret;
-
- ret = btrfs_insert_empty_inode(trans, root, path, ino);
- if (ret)
- return ret;
-
- leaf = path->nodes[0];
- inode_item = btrfs_item_ptr(leaf, path->slots[0],
- struct btrfs_inode_item);
- btrfs_item_key(leaf, &disk_key, path->slots[0]);
- memzero_extent_buffer(leaf, (unsigned long)inode_item,
- sizeof(*inode_item));
- btrfs_set_inode_generation(leaf, inode_item, trans->transid);
- btrfs_set_inode_size(leaf, inode_item, 0);
- btrfs_set_inode_nbytes(leaf, inode_item, 0);
- btrfs_set_inode_uid(leaf, inode_item, 0);
- btrfs_set_inode_gid(leaf, inode_item, 0);
- btrfs_set_inode_mode(leaf, inode_item, S_IFREG | 0600);
- btrfs_set_inode_flags(leaf, inode_item, flags);
- btrfs_set_inode_nlink(leaf, inode_item, 1);
- btrfs_set_inode_transid(leaf, inode_item, trans->transid);
- btrfs_set_inode_block_group(leaf, inode_item, offset);
- btrfs_release_path(path);
-
- key.objectid = BTRFS_FREE_SPACE_OBJECTID;
- key.type = 0;
- key.offset = offset;
- ret = btrfs_insert_empty_item(trans, root, path, &key,
- sizeof(struct btrfs_free_space_header));
- if (ret < 0) {
- btrfs_release_path(path);
- return ret;
- }
-
- leaf = path->nodes[0];
- header = btrfs_item_ptr(leaf, path->slots[0],
- struct btrfs_free_space_header);
- memzero_extent_buffer(leaf, (unsigned long)header, sizeof(*header));
- btrfs_set_free_space_key(leaf, header, &disk_key);
- btrfs_release_path(path);
-
- return 0;
-}
-
-int create_free_space_inode(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path)
-{
- int ret;
- u64 ino;
-
- ret = btrfs_get_free_objectid(trans->fs_info->tree_root, &ino);
- if (ret < 0)
- return ret;
-
- return __create_free_space_inode(trans->fs_info->tree_root, trans, path,
- ino, block_group->start);
-}
-
/*
* inode is an optional sink: if it is NULL, btrfs_remove_free_space_inode
* handles lookup, otherwise it takes ownership and iputs the inode.
@@ -292,7 +195,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans,
}
int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
struct inode *vfs_inode)
{
struct btrfs_truncate_control control = {
@@ -306,33 +208,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans,
struct btrfs_root *root = inode->root;
struct extent_state *cached_state = NULL;
int ret = 0;
- bool locked = false;
-
- if (block_group) {
- BTRFS_PATH_AUTO_FREE(path);
-
- path = btrfs_alloc_path();
- if (!path) {
- ret = -ENOMEM;
- goto fail;
- }
- locked = true;
- mutex_lock(&trans->transaction->cache_write_mutex);
- if (!list_empty(&block_group->io_list)) {
- list_del_init(&block_group->io_list);
-
- btrfs_wait_cache_io(trans, block_group, path);
- btrfs_put_block_group(block_group);
- }
-
- /*
- * now that we've truncated the cache away, its no longer
- * setup or written
- */
- spin_lock(&block_group->lock);
- block_group->disk_cache_state = BTRFS_DC_CLEAR;
- spin_unlock(&block_group->lock);
- }
btrfs_i_size_write(inode, 0);
truncate_pagecache(vfs_inode, 0);
@@ -356,336 +231,12 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans,
ret = btrfs_update_inode(trans, inode);
fail:
- if (locked)
- mutex_unlock(&trans->transaction->cache_write_mutex);
if (ret)
btrfs_abort_transaction(trans, ret);
return ret;
}
-static void readahead_cache(struct inode *inode)
-{
- struct file_ra_state ra;
- pgoff_t last_index;
-
- file_ra_state_init(&ra, inode->i_mapping);
- last_index = (i_size_read(inode) - 1) >> PAGE_SHIFT;
-
- page_cache_sync_readahead(inode->i_mapping, &ra, NULL, 0, last_index);
-}
-
-static int io_ctl_init(struct btrfs_io_ctl *io_ctl, struct inode *inode,
- int write)
-{
- int num_pages;
-
- num_pages = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE);
-
- /* Make sure we can fit our crcs and generation into the first page */
- if (write && (num_pages * sizeof(u32) + sizeof(u64)) > PAGE_SIZE)
- return -ENOSPC;
-
- memset(io_ctl, 0, sizeof(struct btrfs_io_ctl));
-
- io_ctl->pages = kzalloc_objs(struct page *, num_pages, GFP_NOFS);
- if (!io_ctl->pages)
- return -ENOMEM;
-
- io_ctl->num_pages = num_pages;
- io_ctl->fs_info = inode_to_fs_info(inode);
- io_ctl->inode = inode;
-
- return 0;
-}
-ALLOW_ERROR_INJECTION(io_ctl_init, ERRNO);
-
-static void io_ctl_free(struct btrfs_io_ctl *io_ctl)
-{
- kfree(io_ctl->pages);
- io_ctl->pages = NULL;
-}
-
-static void io_ctl_unmap_page(struct btrfs_io_ctl *io_ctl)
-{
- if (io_ctl->cur) {
- io_ctl->cur = NULL;
- io_ctl->orig = NULL;
- }
-}
-
-static void io_ctl_map_page(struct btrfs_io_ctl *io_ctl, int clear)
-{
- ASSERT(io_ctl->index < io_ctl->num_pages);
- io_ctl->page = io_ctl->pages[io_ctl->index++];
- io_ctl->cur = page_address(io_ctl->page);
- io_ctl->orig = io_ctl->cur;
- io_ctl->size = PAGE_SIZE;
- if (clear)
- clear_page(io_ctl->cur);
-}
-
-static void io_ctl_drop_pages(struct btrfs_io_ctl *io_ctl)
-{
- int i;
-
- io_ctl_unmap_page(io_ctl);
-
- for (i = 0; i < io_ctl->num_pages; i++) {
- if (io_ctl->pages[i]) {
- unlock_page(io_ctl->pages[i]);
- put_page(io_ctl->pages[i]);
- }
- }
-}
-
-static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate)
-{
- struct folio *folio;
- struct inode *inode = io_ctl->inode;
- gfp_t mask = btrfs_alloc_write_mask(inode->i_mapping);
- int i;
-
- for (i = 0; i < io_ctl->num_pages; i++) {
- int ret;
-
- folio = __filemap_get_folio(inode->i_mapping, i,
- FGP_LOCK | FGP_ACCESSED | FGP_CREAT,
- mask);
- if (IS_ERR(folio)) {
- io_ctl_drop_pages(io_ctl);
- return PTR_ERR(folio);
- }
-
- ret = set_folio_extent_mapped(folio);
- if (ret < 0) {
- folio_unlock(folio);
- folio_put(folio);
- io_ctl_drop_pages(io_ctl);
- return ret;
- }
-
- io_ctl->pages[i] = &folio->page;
- if (uptodate && !folio_test_uptodate(folio)) {
- btrfs_read_folio(NULL, folio);
- folio_lock(folio);
- if (folio->mapping != inode->i_mapping) {
- btrfs_err(BTRFS_I(inode)->root->fs_info,
- "free space cache page truncated");
- io_ctl_drop_pages(io_ctl);
- return -EIO;
- }
- if (!folio_test_uptodate(folio)) {
- btrfs_err(BTRFS_I(inode)->root->fs_info,
- "error reading free space cache");
- io_ctl_drop_pages(io_ctl);
- return -EIO;
- }
- }
- }
-
- for (i = 0; i < io_ctl->num_pages; i++)
- clear_page_dirty_for_io(io_ctl->pages[i]);
-
- return 0;
-}
-
-static void io_ctl_set_generation(struct btrfs_io_ctl *io_ctl, u64 generation)
-{
- io_ctl_map_page(io_ctl, 1);
-
- /*
- * Skip the csum areas. If we don't check crcs then we just have a
- * 64bit chunk at the front of the first page.
- */
- io_ctl->cur += (sizeof(u32) * io_ctl->num_pages);
- io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages);
-
- put_unaligned_le64(generation, io_ctl->cur);
- io_ctl->cur += sizeof(u64);
-}
-
-static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation)
-{
- u64 cache_gen;
-
- /*
- * Skip the crc area. If we don't check crcs then we just have a 64bit
- * chunk at the front of the first page.
- */
- io_ctl->cur += sizeof(u32) * io_ctl->num_pages;
- io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages);
-
- cache_gen = get_unaligned_le64(io_ctl->cur);
- if (cache_gen != generation) {
- btrfs_err_rl(io_ctl->fs_info,
- "space cache generation (%llu) does not match inode (%llu)",
- cache_gen, generation);
- io_ctl_unmap_page(io_ctl);
- return -EIO;
- }
- io_ctl->cur += sizeof(u64);
- return 0;
-}
-
-static void io_ctl_set_crc(struct btrfs_io_ctl *io_ctl, int index)
-{
- u32 *tmp;
- u32 crc = ~(u32)0;
- unsigned offset = 0;
-
- if (index == 0)
- offset = sizeof(u32) * io_ctl->num_pages;
-
- crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset);
- btrfs_crc32c_final(crc, (u8 *)&crc);
- io_ctl_unmap_page(io_ctl);
- tmp = page_address(io_ctl->pages[0]);
- tmp += index;
- *tmp = crc;
-}
-
-static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index)
-{
- u32 *tmp, val;
- u32 crc = ~(u32)0;
- unsigned offset = 0;
-
- if (index >= io_ctl->num_pages)
- return -EIO;
-
- if (index == 0)
- offset = sizeof(u32) * io_ctl->num_pages;
-
- tmp = page_address(io_ctl->pages[0]);
- tmp += index;
- val = *tmp;
-
- io_ctl_map_page(io_ctl, 0);
- crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset);
- btrfs_crc32c_final(crc, (u8 *)&crc);
- if (val != crc) {
- btrfs_err_rl(io_ctl->fs_info,
- "csum mismatch on free space cache");
- io_ctl_unmap_page(io_ctl);
- return -EIO;
- }
-
- return 0;
-}
-
-static int io_ctl_add_entry(struct btrfs_io_ctl *io_ctl, u64 offset, u64 bytes,
- void *bitmap)
-{
- struct btrfs_free_space_entry *entry;
-
- if (!io_ctl->cur)
- return -ENOSPC;
-
- entry = io_ctl->cur;
- put_unaligned_le64(offset, &entry->offset);
- put_unaligned_le64(bytes, &entry->bytes);
- entry->type = (bitmap) ? BTRFS_FREE_SPACE_BITMAP :
- BTRFS_FREE_SPACE_EXTENT;
- io_ctl->cur += sizeof(struct btrfs_free_space_entry);
- io_ctl->size -= sizeof(struct btrfs_free_space_entry);
-
- if (io_ctl->size >= sizeof(struct btrfs_free_space_entry))
- return 0;
-
- io_ctl_set_crc(io_ctl, io_ctl->index - 1);
-
- /* No more pages to map */
- if (io_ctl->index >= io_ctl->num_pages)
- return 0;
-
- /* map the next page */
- io_ctl_map_page(io_ctl, 1);
- return 0;
-}
-
-static int io_ctl_add_bitmap(struct btrfs_io_ctl *io_ctl, void *bitmap)
-{
- if (!io_ctl->cur)
- return -ENOSPC;
-
- /*
- * If we aren't at the start of the current page, unmap this one and
- * map the next one if there is any left.
- */
- if (io_ctl->cur != io_ctl->orig) {
- io_ctl_set_crc(io_ctl, io_ctl->index - 1);
- if (io_ctl->index >= io_ctl->num_pages)
- return -ENOSPC;
- io_ctl_map_page(io_ctl, 0);
- }
-
- copy_page(io_ctl->cur, bitmap);
- io_ctl_set_crc(io_ctl, io_ctl->index - 1);
- if (io_ctl->index < io_ctl->num_pages)
- io_ctl_map_page(io_ctl, 0);
- return 0;
-}
-
-static void io_ctl_zero_remaining_pages(struct btrfs_io_ctl *io_ctl)
-{
- /*
- * If we're not on the boundary we know we've modified the page and we
- * need to crc the page.
- */
- if (io_ctl->cur != io_ctl->orig)
- io_ctl_set_crc(io_ctl, io_ctl->index - 1);
- else
- io_ctl_unmap_page(io_ctl);
-
- while (io_ctl->index < io_ctl->num_pages) {
- io_ctl_map_page(io_ctl, 1);
- io_ctl_set_crc(io_ctl, io_ctl->index - 1);
- }
-}
-
-static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl,
- struct btrfs_free_space *entry, u8 *type)
-{
- struct btrfs_free_space_entry *e;
- int ret;
-
- if (!io_ctl->cur) {
- ret = io_ctl_check_crc(io_ctl, io_ctl->index);
- if (ret)
- return ret;
- }
-
- e = io_ctl->cur;
- entry->offset = get_unaligned_le64(&e->offset);
- entry->bytes = get_unaligned_le64(&e->bytes);
- *type = e->type;
- io_ctl->cur += sizeof(struct btrfs_free_space_entry);
- io_ctl->size -= sizeof(struct btrfs_free_space_entry);
-
- if (io_ctl->size >= sizeof(struct btrfs_free_space_entry))
- return 0;
-
- io_ctl_unmap_page(io_ctl);
-
- return 0;
-}
-
-static int io_ctl_read_bitmap(struct btrfs_io_ctl *io_ctl,
- struct btrfs_free_space *entry)
-{
- int ret;
-
- ret = io_ctl_check_crc(io_ctl, io_ctl->index);
- if (ret)
- return ret;
-
- copy_page(entry->bitmap, io_ctl->cur);
- io_ctl_unmap_page(io_ctl);
-
- return 0;
-}
-
static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl)
{
struct btrfs_block_group *block_group = ctl->block_group;
@@ -731,824 +282,6 @@ static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl)
div_u64(extent_bytes, sizeof(struct btrfs_free_space));
}
-static int __load_free_space_cache(struct btrfs_root *root, struct inode *inode,
- struct btrfs_free_space_ctl *ctl,
- struct btrfs_path *path, u64 offset)
-{
- struct btrfs_fs_info *fs_info = root->fs_info;
- struct btrfs_free_space_header *header;
- struct extent_buffer *leaf;
- struct btrfs_io_ctl io_ctl;
- struct btrfs_key key;
- struct btrfs_free_space *e, *n;
- LIST_HEAD(bitmaps);
- u64 num_entries;
- u64 num_bitmaps;
- u64 generation;
- u8 type;
- int ret = 0;
-
- /* Nothing in the space cache, goodbye */
- if (!i_size_read(inode))
- return 0;
-
- key.objectid = BTRFS_FREE_SPACE_OBJECTID;
- key.type = 0;
- key.offset = offset;
-
- ret = btrfs_search_slot(NULL, root, &key, path, 0, 0);
- if (ret < 0)
- return 0;
- else if (ret > 0) {
- btrfs_release_path(path);
- return 0;
- }
-
- ret = -1;
-
- leaf = path->nodes[0];
- header = btrfs_item_ptr(leaf, path->slots[0],
- struct btrfs_free_space_header);
- num_entries = btrfs_free_space_entries(leaf, header);
- num_bitmaps = btrfs_free_space_bitmaps(leaf, header);
- generation = btrfs_free_space_generation(leaf, header);
- btrfs_release_path(path);
-
- if (!BTRFS_I(inode)->generation) {
- btrfs_info(fs_info,
- "the free space cache file (%llu) is invalid, skip it",
- offset);
- return 0;
- }
-
- if (BTRFS_I(inode)->generation != generation) {
- btrfs_err(fs_info,
- "free space inode generation (%llu) did not match free space cache generation (%llu)",
- BTRFS_I(inode)->generation, generation);
- return 0;
- }
-
- if (!num_entries)
- return 0;
-
- ret = io_ctl_init(&io_ctl, inode, 0);
- if (ret)
- return ret;
-
- readahead_cache(inode);
-
- ret = io_ctl_prepare_pages(&io_ctl, true);
- if (ret)
- goto out;
-
- ret = io_ctl_check_crc(&io_ctl, 0);
- if (ret)
- goto free_cache;
-
- ret = io_ctl_check_generation(&io_ctl, generation);
- if (ret)
- goto free_cache;
-
- while (num_entries) {
- e = kmem_cache_zalloc(btrfs_free_space_cachep,
- GFP_NOFS);
- if (!e) {
- ret = -ENOMEM;
- goto free_cache;
- }
-
- ret = io_ctl_read_entry(&io_ctl, e, &type);
- if (ret) {
- kmem_cache_free(btrfs_free_space_cachep, e);
- goto free_cache;
- }
-
- if (!e->bytes) {
- ret = -1;
- kmem_cache_free(btrfs_free_space_cachep, e);
- goto free_cache;
- }
-
- if (type == BTRFS_FREE_SPACE_EXTENT) {
- spin_lock(&ctl->tree_lock);
- ret = link_free_space(ctl, e);
- spin_unlock(&ctl->tree_lock);
- if (ret) {
- btrfs_err(fs_info,
- "Duplicate entries in free space cache, dumping");
- kmem_cache_free(btrfs_free_space_cachep, e);
- goto free_cache;
- }
- } else {
- ASSERT(num_bitmaps);
- num_bitmaps--;
- e->bitmap = kmem_cache_zalloc(
- btrfs_free_space_bitmap_cachep, GFP_NOFS);
- if (!e->bitmap) {
- ret = -ENOMEM;
- kmem_cache_free(
- btrfs_free_space_cachep, e);
- goto free_cache;
- }
- spin_lock(&ctl->tree_lock);
- ret = link_free_space(ctl, e);
- if (ret) {
- spin_unlock(&ctl->tree_lock);
- btrfs_err(fs_info,
- "Duplicate entries in free space cache, dumping");
- kmem_cache_free(btrfs_free_space_bitmap_cachep, e->bitmap);
- kmem_cache_free(btrfs_free_space_cachep, e);
- goto free_cache;
- }
- ctl->total_bitmaps++;
- recalculate_thresholds(ctl);
- spin_unlock(&ctl->tree_lock);
- list_add_tail(&e->list, &bitmaps);
- }
-
- num_entries--;
- }
-
- io_ctl_unmap_page(&io_ctl);
-
- /*
- * We add the bitmaps at the end of the entries in order that
- * the bitmap entries are added to the cache.
- */
- list_for_each_entry_safe(e, n, &bitmaps, list) {
- list_del_init(&e->list);
- ret = io_ctl_read_bitmap(&io_ctl, e);
- if (ret)
- goto free_cache;
- }
-
- io_ctl_drop_pages(&io_ctl);
- ret = 1;
-out:
- io_ctl_free(&io_ctl);
- return ret;
-free_cache:
- io_ctl_drop_pages(&io_ctl);
-
- spin_lock(&ctl->tree_lock);
- __btrfs_remove_free_space_cache(ctl);
- spin_unlock(&ctl->tree_lock);
- goto out;
-}
-
-static int copy_free_space_cache(struct btrfs_free_space_ctl *ctl)
-{
- struct btrfs_free_space *info;
- struct rb_node *n;
- int ret = 0;
-
- while (!ret && (n = rb_first(&ctl->free_space_offset)) != NULL) {
- info = rb_entry(n, struct btrfs_free_space, offset_index);
- if (!info->bitmap) {
- const u64 offset = info->offset;
- const u64 bytes = info->bytes;
-
- unlink_free_space(ctl, info, true);
- spin_unlock(&ctl->tree_lock);
- kmem_cache_free(btrfs_free_space_cachep, info);
- ret = btrfs_add_free_space(ctl->block_group, offset, bytes);
- spin_lock(&ctl->tree_lock);
- } else {
- u64 offset = info->offset;
- u64 bytes = ctl->block_group->fs_info->sectorsize;
-
- ret = search_bitmap(ctl, info, &offset, &bytes, false);
- if (ret == 0) {
- bitmap_clear_bits(ctl, info, offset, bytes, true);
- spin_unlock(&ctl->tree_lock);
- ret = btrfs_add_free_space(ctl->block_group, offset,
- bytes);
- spin_lock(&ctl->tree_lock);
- } else {
- free_bitmap(ctl, info);
- ret = 0;
- }
- }
- cond_resched_lock(&ctl->tree_lock);
- }
- return ret;
-}
-
-static struct lock_class_key btrfs_free_space_inode_key;
-
-int load_free_space_cache(struct btrfs_block_group *block_group)
-{
- struct btrfs_fs_info *fs_info = block_group->fs_info;
- struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl;
- struct btrfs_free_space_ctl tmp_ctl = {};
- struct inode *inode;
- struct btrfs_path *path;
- int ret = 0;
- bool matched;
- u64 used = block_group->used;
-
- /*
- * Because we could potentially discard our loaded free space, we want
- * to load everything into a temporary structure first, and then if it's
- * valid copy it all into the actual free space ctl.
- */
- btrfs_init_free_space_ctl(block_group, &tmp_ctl);
-
- /*
- * If this block group has been marked to be cleared for one reason or
- * another then we can't trust the on disk cache, so just return.
- */
- spin_lock(&block_group->lock);
- if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) {
- spin_unlock(&block_group->lock);
- return 0;
- }
- spin_unlock(&block_group->lock);
-
- path = btrfs_alloc_path();
- if (!path)
- return 0;
- path->search_commit_root = true;
- path->skip_locking = true;
-
- /*
- * We must pass a path with search_commit_root set to btrfs_iget in
- * order to avoid a deadlock when allocating extents for the tree root.
- *
- * When we are COWing an extent buffer from the tree root, when looking
- * for a free extent, at extent-tree.c:find_free_extent(), we can find
- * block group without its free space cache loaded. When we find one
- * we must load its space cache which requires reading its free space
- * cache's inode item from the root tree. If this inode item is located
- * in the same leaf that we started COWing before, then we end up in
- * deadlock on the extent buffer (trying to read lock it when we
- * previously write locked it).
- *
- * It's safe to read the inode item using the commit root because
- * block groups, once loaded, stay in memory forever (until they are
- * removed) as well as their space caches once loaded. New block groups
- * once created get their ->cached field set to BTRFS_CACHE_FINISHED so
- * we will never try to read their inode item while the fs is mounted.
- */
- inode = lookup_free_space_inode(block_group, path);
- if (IS_ERR(inode)) {
- btrfs_free_path(path);
- return 0;
- }
-
- /* We may have converted the inode and made the cache invalid. */
- spin_lock(&block_group->lock);
- if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) {
- spin_unlock(&block_group->lock);
- btrfs_free_path(path);
- goto out;
- }
- spin_unlock(&block_group->lock);
-
- /*
- * Reinitialize the class of struct inode's mapping->invalidate_lock for
- * free space inodes to prevent false positives related to locks for normal
- * inodes.
- */
- lockdep_set_class(&(&inode->i_data)->invalidate_lock,
- &btrfs_free_space_inode_key);
-
- ret = __load_free_space_cache(fs_info->tree_root, inode, &tmp_ctl,
- path, block_group->start);
- btrfs_free_path(path);
- if (ret <= 0)
- goto out;
-
- matched = (tmp_ctl.free_space == (block_group->length - used -
- block_group->bytes_super));
-
- if (matched) {
- spin_lock(&tmp_ctl.tree_lock);
- ret = copy_free_space_cache(&tmp_ctl);
- spin_unlock(&tmp_ctl.tree_lock);
- /*
- * ret == 1 means we successfully loaded the free space cache,
- * so we need to re-set it here.
- */
- if (ret == 0)
- ret = 1;
- } else {
- /*
- * We need to call the _locked variant so we don't try to update
- * the discard counters.
- */
- spin_lock(&tmp_ctl.tree_lock);
- __btrfs_remove_free_space_cache(&tmp_ctl);
- spin_unlock(&tmp_ctl.tree_lock);
- btrfs_warn(fs_info,
- "block group %llu has wrong amount of free space",
- block_group->start);
- ret = -1;
- }
-out:
- if (ret < 0) {
- /* This cache is bogus, make sure it gets cleared */
- spin_lock(&block_group->lock);
- block_group->disk_cache_state = BTRFS_DC_CLEAR;
- spin_unlock(&block_group->lock);
- ret = 0;
-
- btrfs_warn(fs_info,
- "failed to load free space cache for block group %llu, rebuilding it now",
- block_group->start);
- }
-
- spin_lock(&ctl->tree_lock);
- btrfs_discard_update_discardable(block_group);
- spin_unlock(&ctl->tree_lock);
- iput(inode);
- return ret;
-}
-
-static noinline_for_stack
-int write_cache_extent_entries(struct btrfs_io_ctl *io_ctl,
- struct btrfs_block_group *block_group,
- int *entries, int *bitmaps,
- struct list_head *bitmap_list)
-{
- int ret;
- struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl;
- struct btrfs_free_cluster *cluster = NULL;
- struct btrfs_free_cluster *cluster_locked = NULL;
- struct rb_node *node = rb_first(&ctl->free_space_offset);
- struct btrfs_trim_range *trim_entry;
-
- /* Get the cluster for this block_group if it exists */
- if (!list_empty(&block_group->cluster_list)) {
- cluster = list_first_entry(&block_group->cluster_list,
- struct btrfs_free_cluster, block_group_list);
- }
-
- if (!node && cluster) {
- cluster_locked = cluster;
- spin_lock(&cluster_locked->lock);
- node = rb_first(&cluster->root);
- cluster = NULL;
- }
-
- /* Write out the extent entries */
- while (node) {
- struct btrfs_free_space *e;
-
- e = rb_entry(node, struct btrfs_free_space, offset_index);
- *entries += 1;
-
- ret = io_ctl_add_entry(io_ctl, e->offset, e->bytes,
- e->bitmap);
- if (ret)
- goto fail;
-
- if (e->bitmap) {
- list_add_tail(&e->list, bitmap_list);
- *bitmaps += 1;
- }
- node = rb_next(node);
- if (!node && cluster) {
- node = rb_first(&cluster->root);
- cluster_locked = cluster;
- spin_lock(&cluster_locked->lock);
- cluster = NULL;
- }
- }
- if (cluster_locked) {
- spin_unlock(&cluster_locked->lock);
- cluster_locked = NULL;
- }
-
- /*
- * Make sure we don't miss any range that was removed from our rbtree
- * because trimming is running. Otherwise after a umount+mount (or crash
- * after committing the transaction) we would leak free space and get
- * an inconsistent free space cache report from fsck.
- */
- list_for_each_entry(trim_entry, &ctl->trimming_ranges, list) {
- ret = io_ctl_add_entry(io_ctl, trim_entry->start,
- trim_entry->bytes, NULL);
- if (ret)
- goto fail;
- *entries += 1;
- }
-
- return 0;
-fail:
- if (cluster_locked)
- spin_unlock(&cluster_locked->lock);
- return -ENOSPC;
-}
-
-static noinline_for_stack int
-update_cache_item(struct btrfs_trans_handle *trans,
- struct btrfs_root *root,
- struct inode *inode,
- struct btrfs_path *path, u64 offset,
- int entries, int bitmaps)
-{
- struct btrfs_key key;
- struct btrfs_free_space_header *header;
- struct extent_buffer *leaf;
- int ret;
-
- key.objectid = BTRFS_FREE_SPACE_OBJECTID;
- key.type = 0;
- key.offset = offset;
-
- ret = btrfs_search_slot(trans, root, &key, path, 0, 1);
- if (ret < 0) {
- btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1,
- EXTENT_DELALLOC, NULL);
- return ret;
- }
- leaf = path->nodes[0];
- if (ret > 0) {
- struct btrfs_key found_key;
- ASSERT(path->slots[0]);
- path->slots[0]--;
- btrfs_item_key_to_cpu(leaf, &found_key, path->slots[0]);
- if (found_key.objectid != BTRFS_FREE_SPACE_OBJECTID ||
- found_key.offset != offset) {
- btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0,
- inode->i_size - 1, EXTENT_DELALLOC,
- NULL);
- btrfs_release_path(path);
- return -ENOENT;
- }
- }
-
- BTRFS_I(inode)->generation = trans->transid;
- header = btrfs_item_ptr(leaf, path->slots[0],
- struct btrfs_free_space_header);
- btrfs_set_free_space_entries(leaf, header, entries);
- btrfs_set_free_space_bitmaps(leaf, header, bitmaps);
- btrfs_set_free_space_generation(leaf, header, trans->transid);
- btrfs_release_path(path);
-
- return 0;
-}
-
-static noinline_for_stack int write_pinned_extent_entries(
- struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_io_ctl *io_ctl,
- int *entries)
-{
- u64 start, extent_start, extent_end, len;
- const u64 block_group_end = btrfs_block_group_end(block_group);
- struct extent_io_tree *unpin = NULL;
- int ret;
-
- /*
- * We want to add any pinned extents to our free space cache
- * so we don't leak the space
- *
- * We shouldn't have switched the pinned extents yet so this is the
- * right one
- */
- unpin = &trans->transaction->pinned_extents;
-
- start = block_group->start;
-
- while (start < block_group_end) {
- if (!btrfs_find_first_extent_bit(unpin, start,
- &extent_start, &extent_end,
- EXTENT_DIRTY, NULL))
- return 0;
-
- /* This pinned extent is out of our range */
- if (extent_start >= block_group_end)
- return 0;
-
- extent_start = max(extent_start, start);
- extent_end = min(block_group_end, extent_end + 1);
- len = extent_end - extent_start;
-
- *entries += 1;
- ret = io_ctl_add_entry(io_ctl, extent_start, len, NULL);
- if (ret)
- return -ENOSPC;
-
- start = extent_end;
- }
-
- return 0;
-}
-
-static noinline_for_stack int
-write_bitmap_entries(struct btrfs_io_ctl *io_ctl, struct list_head *bitmap_list)
-{
- struct btrfs_free_space *entry, *next;
- int ret;
-
- /* Write out the bitmaps */
- list_for_each_entry_safe(entry, next, bitmap_list, list) {
- ret = io_ctl_add_bitmap(io_ctl, entry->bitmap);
- if (ret)
- return -ENOSPC;
- list_del_init(&entry->list);
- }
-
- return 0;
-}
-
-static int flush_dirty_cache(struct inode *inode)
-{
- int ret;
-
- ret = btrfs_wait_ordered_range(BTRFS_I(inode), 0, (u64)-1);
- if (ret)
- btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1,
- EXTENT_DELALLOC, NULL);
-
- return ret;
-}
-
-static void noinline_for_stack
-cleanup_bitmap_list(struct list_head *bitmap_list)
-{
- struct btrfs_free_space *entry, *next;
-
- list_for_each_entry_safe(entry, next, bitmap_list, list)
- list_del_init(&entry->list);
-}
-
-static void noinline_for_stack
-cleanup_write_cache_enospc(struct inode *inode,
- struct btrfs_io_ctl *io_ctl,
- struct extent_state **cached_state)
-{
- io_ctl_drop_pages(io_ctl);
- btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1,
- cached_state);
-}
-
-static int __btrfs_wait_cache_io(struct btrfs_root *root,
- struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_io_ctl *io_ctl,
- struct btrfs_path *path, u64 offset)
-{
- int ret;
- struct inode *inode = io_ctl->inode;
-
- if (!inode)
- return 0;
-
- /* Flush the dirty pages in the cache file. */
- ret = flush_dirty_cache(inode);
- if (ret)
- goto out;
-
- /* Update the cache item to tell everyone this cache file is valid. */
- ret = update_cache_item(trans, root, inode, path, offset,
- io_ctl->entries, io_ctl->bitmaps);
-out:
- if (ret) {
- invalidate_inode_pages2(inode->i_mapping);
- BTRFS_I(inode)->generation = 0;
- if (block_group)
- btrfs_debug(root->fs_info,
- "failed to write free space cache for block group %llu error %d",
- block_group->start, ret);
- }
- btrfs_update_inode(trans, BTRFS_I(inode));
-
- if (block_group) {
- /* the dirty list is protected by the dirty_bgs_lock */
- spin_lock(&trans->transaction->dirty_bgs_lock);
-
- /* the disk_cache_state is protected by the block group lock */
- spin_lock(&block_group->lock);
-
- /*
- * only mark this as written if we didn't get put back on
- * the dirty list while waiting for IO. Otherwise our
- * cache state won't be right, and we won't get written again
- */
- if (!ret && list_empty(&block_group->dirty_list))
- block_group->disk_cache_state = BTRFS_DC_WRITTEN;
- else if (ret)
- block_group->disk_cache_state = BTRFS_DC_ERROR;
-
- spin_unlock(&block_group->lock);
- spin_unlock(&trans->transaction->dirty_bgs_lock);
- io_ctl->inode = NULL;
- iput(inode);
- }
-
- return ret;
-
-}
-
-int btrfs_wait_cache_io(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path)
-{
- return __btrfs_wait_cache_io(block_group->fs_info->tree_root, trans,
- block_group, &block_group->io_ctl,
- path, block_group->start);
-}
-
-/*
- * Write out cached info to an inode.
- *
- * @inode: freespace inode we are writing out
- * @ctl: free space cache we are going to write out
- * @block_group: block_group for this cache if it belongs to a block_group
- * @io_ctl: holds context for the io
- * @trans: the trans handle
- *
- * This function writes out a free space cache struct to disk for quick recovery
- * on mount. This will return 0 if it was successful in writing the cache out,
- * or an errno if it was not.
- */
-static int __btrfs_write_out_cache(struct inode *inode,
- struct btrfs_block_group *block_group,
- struct btrfs_trans_handle *trans)
-{
- struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl;
- struct btrfs_io_ctl *io_ctl = &block_group->io_ctl;
- struct extent_state *cached_state = NULL;
- LIST_HEAD(bitmap_list);
- int entries = 0;
- int bitmaps = 0;
- int ret;
- bool must_iput = false;
- int i_size;
-
- if (!i_size_read(inode))
- return -EIO;
-
- WARN_ON(io_ctl->pages);
- ret = io_ctl_init(io_ctl, inode, 1);
- if (ret)
- return ret;
-
- if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) {
- down_write(&block_group->data_rwsem);
- spin_lock(&block_group->lock);
- if (block_group->delalloc_bytes) {
- block_group->disk_cache_state = BTRFS_DC_WRITTEN;
- spin_unlock(&block_group->lock);
- up_write(&block_group->data_rwsem);
- BTRFS_I(inode)->generation = 0;
- ret = 0;
- must_iput = true;
- goto out;
- }
- spin_unlock(&block_group->lock);
- }
-
- /* Lock all pages first so we can lock the extent safely. */
- ret = io_ctl_prepare_pages(io_ctl, false);
- if (ret)
- goto out_unlock;
-
- btrfs_lock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1,
- &cached_state);
-
- io_ctl_set_generation(io_ctl, trans->transid);
-
- mutex_lock(&ctl->cache_writeout_mutex);
- /* Write out the extent entries in the free space cache */
- spin_lock(&ctl->tree_lock);
- ret = write_cache_extent_entries(io_ctl, block_group, &entries, &bitmaps,
- &bitmap_list);
- if (ret)
- goto out_nospc_locked;
-
- /*
- * Some spaces that are freed in the current transaction are pinned,
- * they will be added into free space cache after the transaction is
- * committed, we shouldn't lose them.
- *
- * If this changes while we are working we'll get added back to
- * the dirty list and redo it. No locking needed
- */
- ret = write_pinned_extent_entries(trans, block_group, io_ctl, &entries);
- if (ret)
- goto out_nospc_locked;
-
- /*
- * At last, we write out all the bitmaps and keep cache_writeout_mutex
- * locked while doing it because a concurrent trim can be manipulating
- * or freeing the bitmap.
- */
- ret = write_bitmap_entries(io_ctl, &bitmap_list);
- spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
- if (ret)
- goto out_nospc;
-
- /* Zero out the rest of the pages just to make sure */
- io_ctl_zero_remaining_pages(io_ctl);
-
- /* Everything is written out, now we dirty the pages in the file. */
- i_size = i_size_read(inode);
- for (int i = 0; i < round_up(i_size, PAGE_SIZE) / PAGE_SIZE; i++) {
- u64 dirty_start = i * PAGE_SIZE;
- u64 dirty_len = min_t(u64, dirty_start + PAGE_SIZE, i_size) - dirty_start;
-
- ret = btrfs_dirty_folio(BTRFS_I(inode), page_folio(io_ctl->pages[i]),
- dirty_start, dirty_len, &cached_state, false);
- if (ret < 0)
- goto out_nospc;
- }
-
- if (block_group->flags & BTRFS_BLOCK_GROUP_DATA)
- up_write(&block_group->data_rwsem);
- /*
- * Release the pages and unlock the extent, we will flush
- * them out later
- */
- io_ctl_drop_pages(io_ctl);
- io_ctl_free(io_ctl);
-
- btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1,
- &cached_state);
-
- /*
- * at this point the pages are under IO and we're happy,
- * The caller is responsible for waiting on them and updating
- * the cache and the inode
- */
- io_ctl->entries = entries;
- io_ctl->bitmaps = bitmaps;
-
- ret = btrfs_fdatawrite_range(BTRFS_I(inode), 0, (u64)-1);
- if (ret)
- goto out;
-
- return 0;
-
-out_nospc_locked:
- cleanup_bitmap_list(&bitmap_list);
- spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
-
-out_nospc:
- cleanup_write_cache_enospc(inode, io_ctl, &cached_state);
-
-out_unlock:
- if (block_group->flags & BTRFS_BLOCK_GROUP_DATA)
- up_write(&block_group->data_rwsem);
-
-out:
- io_ctl->inode = NULL;
- io_ctl_free(io_ctl);
- if (ret) {
- invalidate_inode_pages2(inode->i_mapping);
- BTRFS_I(inode)->generation = 0;
- }
- btrfs_update_inode(trans, BTRFS_I(inode));
- if (must_iput)
- iput(inode);
- return ret;
-}
-
-int btrfs_write_out_cache(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path)
-{
- struct btrfs_fs_info *fs_info = trans->fs_info;
- struct inode *inode;
- int ret = 0;
-
- spin_lock(&block_group->lock);
- if (block_group->disk_cache_state < BTRFS_DC_SETUP) {
- spin_unlock(&block_group->lock);
- return 0;
- }
- spin_unlock(&block_group->lock);
-
- inode = lookup_free_space_inode(block_group, path);
- if (IS_ERR(inode))
- return 0;
-
- ret = __btrfs_write_out_cache(inode, block_group, trans);
- if (ret) {
- btrfs_debug(fs_info,
- "failed to write free space cache for block group %llu error %d",
- block_group->start, ret);
- spin_lock(&block_group->lock);
- block_group->disk_cache_state = BTRFS_DC_ERROR;
- spin_unlock(&block_group->lock);
-
- block_group->io_ctl.inode = NULL;
- iput(inode);
- }
-
- /*
- * if ret == 0 the caller is expected to call btrfs_wait_cache_io
- * to wait for IO and put the inode
- */
-
- return ret;
-}
-
static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit,
u64 offset)
{
@@ -2953,8 +1686,6 @@ void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group,
spin_lock_init(&ctl->tree_lock);
ctl->block_group = block_group;
ctl->free_space_bytes = RB_ROOT_CACHED;
- INIT_LIST_HEAD(&ctl->trimming_ranges);
- mutex_init(&ctl->cache_writeout_mutex);
/*
* we only want to have 32k of ram per block group for keeping
@@ -3650,12 +2381,10 @@ void btrfs_init_free_cluster(struct btrfs_free_cluster *cluster)
static int do_trimming(struct btrfs_block_group *block_group,
u64 *total_trimmed, u64 start, u64 bytes,
u64 reserved_start, u64 reserved_bytes,
- enum btrfs_trim_state reserved_trim_state,
- struct btrfs_trim_range *trim_entry)
+ enum btrfs_trim_state reserved_trim_state)
{
struct btrfs_space_info *space_info = block_group->space_info;
struct btrfs_fs_info *fs_info = block_group->fs_info;
- struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl;
int ret;
bool bg_ro;
const u64 end = start + bytes;
@@ -3681,7 +2410,6 @@ static int do_trimming(struct btrfs_block_group *block_group,
trim_state = BTRFS_TRIM_STATE_TRIMMED;
}
- mutex_lock(&ctl->cache_writeout_mutex);
if (reserved_start < start)
__btrfs_add_free_space(block_group, reserved_start,
start - reserved_start,
@@ -3690,8 +2418,6 @@ static int do_trimming(struct btrfs_block_group *block_group,
__btrfs_add_free_space(block_group, end, reserved_end - end,
reserved_trim_state);
__btrfs_add_free_space(block_group, start, bytes, trim_state);
- list_del(&trim_entry->list);
- mutex_unlock(&ctl->cache_writeout_mutex);
if (!bg_ro) {
spin_lock(&space_info->lock);
@@ -3729,9 +2455,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group,
const u64 max_discard_size = READ_ONCE(discard_ctl->max_discard_size);
while (start < end) {
- struct btrfs_trim_range trim_entry;
-
- mutex_lock(&ctl->cache_writeout_mutex);
spin_lock(&ctl->tree_lock);
if (ctl->free_space < minlen)
@@ -3762,7 +2485,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group,
bytes = entry->bytes;
if (bytes < minlen) {
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
goto next;
}
unlink_free_space(ctl, entry, true);
@@ -3787,7 +2509,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group,
bytes = min(extent_start + extent_bytes, end) - start;
if (bytes < minlen) {
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
goto next;
}
@@ -3796,14 +2517,9 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group,
}
spin_unlock(&ctl->tree_lock);
- trim_entry.start = extent_start;
- trim_entry.bytes = extent_bytes;
- list_add_tail(&trim_entry.list, &ctl->trimming_ranges);
- mutex_unlock(&ctl->cache_writeout_mutex);
ret = do_trimming(block_group, total_trimmed, start, bytes,
- extent_start, extent_bytes, extent_trim_state,
- &trim_entry);
+ extent_start, extent_bytes, extent_trim_state);
if (ret) {
block_group->discard_cursor = start + bytes;
break;
@@ -3827,7 +2543,6 @@ next:
out_unlock:
block_group->discard_cursor = btrfs_block_group_end(block_group);
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
return ret;
}
@@ -3938,16 +2653,13 @@ static int trim_bitmaps(struct btrfs_block_group *block_group,
while (offset < end) {
bool next_bitmap = false;
- struct btrfs_trim_range trim_entry;
- mutex_lock(&ctl->cache_writeout_mutex);
spin_lock(&ctl->tree_lock);
if (ctl->free_space < minlen) {
block_group->discard_cursor =
btrfs_block_group_end(block_group);
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
break;
}
@@ -3963,7 +2675,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group,
if (!entry || (async && minlen && start == offset &&
btrfs_free_space_trimmed(entry))) {
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
next_bitmap = true;
goto next;
}
@@ -3989,7 +2700,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group,
else
entry->trim_state = BTRFS_TRIM_STATE_UNTRIMMED;
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
next_bitmap = true;
goto next;
}
@@ -4000,14 +2710,12 @@ static int trim_bitmaps(struct btrfs_block_group *block_group,
*/
if (async && *total_trimmed) {
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
return ret;
}
bytes = min(bytes, end - start);
if (bytes < minlen || (async && maxlen && bytes > maxlen)) {
spin_unlock(&ctl->tree_lock);
- mutex_unlock(&ctl->cache_writeout_mutex);
goto next;
}
@@ -4027,13 +2735,9 @@ static int trim_bitmaps(struct btrfs_block_group *block_group,
free_bitmap(ctl, entry);
spin_unlock(&ctl->tree_lock);
- trim_entry.start = start;
- trim_entry.bytes = bytes;
- list_add_tail(&trim_entry.list, &ctl->trimming_ranges);
- mutex_unlock(&ctl->cache_writeout_mutex);
ret = do_trimming(block_group, total_trimmed, start, bytes,
- start, bytes, 0, &trim_entry);
+ start, bytes, 0);
if (ret) {
reset_trimming_bitmap(ctl, offset);
block_group->discard_cursor =
@@ -4152,47 +2856,29 @@ bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info)
return btrfs_super_cache_generation(fs_info->super_copy);
}
-static int cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info,
- struct btrfs_trans_handle *trans)
+int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info)
{
- struct btrfs_block_group *block_group;
+ struct btrfs_trans_handle *trans;
struct rb_node *node;
+ int ret;
btrfs_info(fs_info, "cleaning free space cache v1");
- node = rb_first_cached(&fs_info->block_group_cache_tree);
- while (node) {
- int ret;
-
- block_group = rb_entry(node, struct btrfs_block_group, cache_node);
- ret = btrfs_remove_free_space_inode(trans, NULL, block_group);
- if (ret)
- return ret;
- node = rb_next(node);
- }
- return 0;
-}
-
-int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active)
-{
- struct btrfs_trans_handle *trans;
- int ret;
-
/*
- * update_super_roots will appropriately set or unset
- * super_copy->cache_generation based on SPACE_CACHE and
- * BTRFS_FS_CLEANUP_SPACE_CACHE_V1. For this reason, we need a
- * transaction commit whether we are enabling space cache v1 and don't
- * have any other work to do, or are disabling it and removing free
- * space inodes.
+ * update_super_roots() zeroes super_copy->cache_generation while
+ * BTRFS_FS_CLEANUP_SPACE_CACHE_V1 is set, so this needs a commit.
*/
trans = btrfs_start_transaction(fs_info->tree_root, 0);
if (IS_ERR(trans))
return PTR_ERR(trans);
- if (!active) {
- set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags);
- ret = cleanup_free_space_cache_v1(fs_info, trans);
+ set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags);
+ for (node = rb_first_cached(&fs_info->block_group_cache_tree); node;
+ node = rb_next(node)) {
+ struct btrfs_block_group *block_group;
+
+ block_group = rb_entry(node, struct btrfs_block_group, cache_node);
+ ret = btrfs_remove_free_space_inode(trans, NULL, block_group);
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
btrfs_end_transaction(trans);
diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h
index 53fe8e293af1..e22443598b8e 100644
--- a/fs/btrfs/free-space-cache.h
+++ b/fs/btrfs/free-space-cache.h
@@ -14,7 +14,6 @@
#include "fs.h"
struct inode;
-struct page;
struct btrfs_fs_info;
struct btrfs_path;
struct btrfs_trans_handle;
@@ -84,44 +83,18 @@ struct btrfs_free_space_ctl {
s32 discardable_extents[BTRFS_STAT_NR_ENTRIES];
s64 discardable_bytes[BTRFS_STAT_NR_ENTRIES];
struct btrfs_block_group *block_group;
- struct mutex cache_writeout_mutex;
- struct list_head trimming_ranges;
-};
-
-struct btrfs_io_ctl {
- void *cur, *orig;
- struct page *page;
- struct page **pages;
- struct btrfs_fs_info *fs_info;
- struct inode *inode;
- unsigned long size;
- int index;
- int num_pages;
- int entries;
- int bitmaps;
};
int __init btrfs_free_space_init(void);
void __cold btrfs_free_space_exit(void);
struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group,
struct btrfs_path *path);
-int create_free_space_inode(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path);
int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans,
struct inode *inode,
struct btrfs_block_group *block_group);
int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
struct inode *inode);
-int load_free_space_cache(struct btrfs_block_group *block_group);
-int btrfs_wait_cache_io(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path);
-int btrfs_write_out_cache(struct btrfs_trans_handle *trans,
- struct btrfs_block_group *block_group,
- struct btrfs_path *path);
void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group,
struct btrfs_free_space_ctl *ctl);
@@ -161,7 +134,7 @@ int btrfs_trim_block_group_bitmaps(struct btrfs_block_group *block_group,
void btrfs_trim_fully_remapped_block_group(struct btrfs_block_group *bg);
bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info);
-int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active);
+int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info);
/* Support functions for running our sanity tests */
#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS
bool btrfs_use_bitmap(struct btrfs_free_space_ctl *ctl,
diff --git a/fs/btrfs/fs.c b/fs/btrfs/fs.c
index de160d29dde8..75a1217727a7 100644
--- a/fs/btrfs/fs.c
+++ b/fs/btrfs/fs.c
@@ -79,7 +79,7 @@ void btrfs_csum_init(struct btrfs_csum_ctx *ctx, u16 csum_type)
blake2b_init(&ctx->blake2b, 32);
break;
default:
- /* Checksume type is validated at mount time. */
+ /* Checksum type is validated at mount time. */
BUG();
}
}
diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h
index 10e15a319b93..79d0828c51c7 100644
--- a/fs/btrfs/fs.h
+++ b/fs/btrfs/fs.h
@@ -259,7 +259,6 @@ enum {
BTRFS_MOUNT_NOSSD = (1ULL << 9),
BTRFS_MOUNT_DISCARD_SYNC = (1ULL << 10),
BTRFS_MOUNT_FORCE_COMPRESS = (1ULL << 11),
- BTRFS_MOUNT_SPACE_CACHE = (1ULL << 12),
BTRFS_MOUNT_CLEAR_CACHE = (1ULL << 13),
BTRFS_MOUNT_USER_SUBVOL_RM_ALLOWED = (1ULL << 14),
BTRFS_MOUNT_ENOSPC_DEBUG = (1ULL << 15),
@@ -712,7 +711,6 @@ struct btrfs_fs_info {
struct workqueue_struct *endio_meta_workers;
struct workqueue_struct *rmw_workers;
struct btrfs_workqueue *endio_write_workers;
- struct btrfs_workqueue *endio_freespace_worker;
struct btrfs_workqueue *caching_workers;
struct workqueue_struct *fixup_workers;
@@ -811,7 +809,7 @@ struct btrfs_fs_info {
struct btrfs_discard_ctl discard_ctl;
/* Is qgroup tracking in a consistent state? */
- u64 qgroup_flags;
+ unsigned long qgroup_flags;
/* Holds configuration and tracking. Protected by qgroup_lock. */
struct rb_root qgroup_tree;
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index 558b4a3f9633..f4b68205f621 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -730,6 +730,9 @@ static inline int inode_need_compress(struct btrfs_inode *inode, u64 start,
u64 end, bool check_inline)
{
struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ const u32 blocksize = fs_info->sectorsize;
+
+ ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize));
if (unlikely(!btrfs_inode_can_compress(inode))) {
DEBUG_WARN("BTRFS: unexpected compression for ino %llu", btrfs_ino(inode));
@@ -1366,11 +1369,6 @@ static noinline int cow_file_range(struct btrfs_inode *inode,
goto out_unlock;
}
- if (btrfs_is_free_space_inode(inode)) {
- ret = -EINVAL;
- goto out_unlock;
- }
-
num_bytes = ALIGN(end - start + 1, blocksize);
num_bytes = max(blocksize, num_bytes);
ASSERT(num_bytes <= btrfs_super_total_bytes(fs_info->super_copy));
@@ -1680,7 +1678,6 @@ static int fallback_to_cow(struct btrfs_inode *inode,
struct folio *locked_folio, const u64 start,
const u64 end)
{
- const bool is_space_ino = btrfs_is_free_space_inode(inode);
const bool is_reloc_ino = btrfs_is_data_reloc_root(inode->root);
const u64 range_bytes = end + 1 - start;
struct extent_io_tree *io_tree = &inode->io_tree;
@@ -1713,23 +1710,22 @@ static int fallback_to_cow(struct btrfs_inode *inode,
* extent_clear_unlock_delalloc()) the bytes_may_use counter of the
* data space info, which we incremented in the step above.
*
- * If we need to fallback to cow and the inode corresponds to a free
- * space cache inode or an inode of the data relocation tree, we must
- * also increment bytes_may_use of the data space_info for the same
- * reason. Space caches and relocated data extents always get a prealloc
- * extent for them, however scrub or balance may have set the block
- * group that contains that extent to RO mode and therefore force COW
- * when starting writeback.
+ * If we need to fallback to cow and the inode is in the data relocation
+ * tree, we must also increment bytes_may_use of the data space_info for
+ * the same reason. Relocated data extents always get a prealloc extent,
+ * however scrub or balance may have set the block group that contains
+ * that extent to RO mode and therefore force COW when starting
+ * writeback.
*/
btrfs_lock_extent(io_tree, start, end, &cached_state);
count = btrfs_count_range_bits(io_tree, &range_start, end, range_bytes,
EXTENT_NORESERVE, false, NULL);
- if (count > 0 || is_space_ino || is_reloc_ino) {
+ if (count > 0 || is_reloc_ino) {
u64 bytes = count;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
struct btrfs_space_info *sinfo = fs_info->data_sinfo;
- if (is_space_ino || is_reloc_ino)
+ if (is_reloc_ino)
bytes = range_bytes;
spin_lock(&sinfo->lock);
@@ -1794,7 +1790,6 @@ static int can_nocow_file_extent(struct btrfs_path *path,
struct btrfs_inode *inode,
struct can_nocow_file_extent_args *args)
{
- const bool is_freespace_inode = btrfs_is_free_space_inode(inode);
struct extent_buffer *leaf = path->nodes[0];
struct btrfs_root *root = inode->root;
struct btrfs_file_extent_item *fi;
@@ -1807,8 +1802,7 @@ static int can_nocow_file_extent(struct btrfs_path *path,
bool nowait = path->nowait;
/* If there are pending snapshots for this root, we must do COW. */
- if (args->writeback_path && !is_freespace_inode &&
- atomic_read(&root->snapshot_force_cow))
+ if (args->writeback_path && atomic_read(&root->snapshot_force_cow))
goto out;
fi = btrfs_item_ptr(leaf, path->slots[0], struct btrfs_file_extent_item);
@@ -1857,7 +1851,6 @@ static int can_nocow_file_extent(struct btrfs_path *path,
ret = btrfs_cross_ref_exist(inode, key->offset - args->file_extent.offset,
args->file_extent.disk_bytenr, path);
- WARN_ON_ONCE(ret > 0 && is_freespace_inode);
if (ret != 0)
goto out;
@@ -1892,7 +1885,6 @@ static int can_nocow_file_extent(struct btrfs_path *path,
ret = btrfs_lookup_csums_list(csum_root, io_start,
io_start + args->file_extent.num_bytes - 1,
NULL, nowait);
- WARN_ON_ONCE(ret > 0 && is_freespace_inode);
if (ret != 0)
goto out;
@@ -2331,7 +2323,7 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f
btrfs_check_folio_write_protected(locked_folio);
if (btrfs_inode_can_compress(inode) &&
- inode_need_compress(inode, 0, blocksize, true)) {
+ inode_need_compress(inode, 0, blocksize - 1, true)) {
if (inode->defrag_compress > 0 &&
inode->defrag_compress < BTRFS_NR_COMPRESS_TYPES) {
compress_type = inode->defrag_compress;
@@ -2640,7 +2632,7 @@ void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *s
* and are therefore protected against concurrent calls of this
* function and btrfs_clear_delalloc_extent().
*/
- if (!btrfs_is_free_space_inode(inode) && prev_delalloc_bytes == 0)
+ if (prev_delalloc_bytes == 0)
btrfs_add_delalloc_inode(inode);
}
@@ -2698,7 +2690,6 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode,
return;
if (!btrfs_is_data_reloc_root(root) &&
- !btrfs_is_free_space_inode(inode) &&
!(state->state & EXTENT_NORESERVE) &&
(bits & EXTENT_CLEAR_DATA_RESV))
btrfs_free_reserved_data_space_noquota(inode, len);
@@ -2716,7 +2707,7 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode,
* and are therefore protected against concurrent calls of this
* function and btrfs_set_delalloc_extent().
*/
- if (!btrfs_is_free_space_inode(inode) && new_delalloc_bytes == 0) {
+ if (new_delalloc_bytes == 0) {
spin_lock(&root->delalloc_lock);
btrfs_del_delalloc_inode(inode);
spin_unlock(&root->delalloc_lock);
@@ -3218,7 +3209,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent)
int compress_type = 0;
int ret = 0;
u64 logical_len = ordered_extent->num_bytes;
- bool freespace_inode;
+ u64 unwritten_start;
bool truncated = false;
bool clear_reserved_extent = true;
unsigned int clear_bits = 0;
@@ -3235,9 +3226,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent)
if (!test_bit(BTRFS_ORDERED_NOCOW, &ordered_extent->flags))
clear_bits |= EXTENT_DEFRAG;
- freespace_inode = btrfs_is_free_space_inode(inode);
- if (!freespace_inode)
- btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent);
+ btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent);
if (unlikely(test_bit(BTRFS_ORDERED_IOERR, &ordered_extent->flags))) {
ret = -EIO;
@@ -3272,10 +3261,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent)
&cached_state);
}
- if (freespace_inode)
- trans = btrfs_join_transaction_spacecache(root);
- else
- trans = btrfs_join_transaction(root);
+ trans = btrfs_join_transaction(root);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
trans = NULL;
@@ -3385,29 +3371,11 @@ out:
if (ret)
btrfs_mark_ordered_extent_error(ordered_extent);
- /*
- * Drop extent maps for the part of the extent we didn't write.
- *
- * We have an exception here for the free_space_inode, this is
- * because when we do btrfs_get_extent() on the free space inode
- * we will search the commit root. If this is a new block group
- * we won't find anything, and we will trip over the assert in
- * writepage where we do ASSERT(em->block_start !=
- * EXTENT_MAP_HOLE).
- *
- * Theoretically we could also skip this for any NOCOW extent as
- * we don't mess with the extent map tree in the NOCOW case, but
- * for now simply skip this if we are the free space inode.
- */
- if (!btrfs_is_free_space_inode(inode)) {
- u64 unwritten_start = start;
-
- if (truncated)
- unwritten_start += logical_len;
-
- btrfs_drop_extent_map_range(inode, unwritten_start,
- end, false);
- }
+ /* Drop extent maps for the part of the extent we didn't write. */
+ unwritten_start = start;
+ if (truncated)
+ unwritten_start += logical_len;
+ btrfs_drop_extent_map_range(inode, unwritten_start, end, false);
/*
* If the ordered extent had an IOERR or something else went
@@ -3471,79 +3439,30 @@ int btrfs_finish_ordered_io(struct btrfs_ordered_extent *ordered)
return btrfs_finish_one_ordered(ordered);
}
-/*
- * Calculate the checksum of an fs block at physical memory address @paddr,
- * and save the result to @dest.
- *
- * The folio containing @paddr must be large enough to contain a full fs block.
- */
-void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info,
- const phys_addr_t paddr, u8 *dest)
+/* Generate data checksum for a single fs block, pointed to by @orig_iter. */
+void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio,
+ const struct bvec_iter *orig_iter, u8 *csum)
{
- struct folio *folio = page_folio(phys_to_page(paddr));
+ struct btrfs_csum_ctx cctx;
+ struct bvec_iter iter = *orig_iter;
const u32 blocksize = fs_info->sectorsize;
- const u32 step = min(blocksize, PAGE_SIZE);
- const u32 nr_steps = blocksize / step;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
-
- /* The full block must be inside the folio. */
- ASSERT(offset_in_folio(folio, paddr) + blocksize <= folio_size(folio));
+ u32 cur = 0;
- for (int i = 0; i < nr_steps; i++) {
- u32 pindex = offset_in_folio(folio, paddr + i * step) >> PAGE_SHIFT;
-
- /*
- * For bs <= ps cases, we will only run the loop once, so the offset
- * inside the page will only added to paddrs[0].
- *
- * For bs > ps cases, the block must be page aligned, thus offset
- * inside the page will always be 0.
- */
- paddrs[i] = page_to_phys(folio_page(folio, pindex)) + offset_in_page(paddr);
- }
- return btrfs_calculate_block_csum_pages(fs_info, paddrs, dest);
-}
-
-/*
- * Calculate the checksum of a fs block backed by multiple noncontiguous pages
- * at @paddrs[] and save the result to @dest.
- *
- * The folio containing @paddr must be large enough to contain a full fs block.
- */
-void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info,
- const phys_addr_t paddrs[], u8 *dest)
-{
- const u32 blocksize = fs_info->sectorsize;
- const u32 step = min(blocksize, PAGE_SIZE);
- const u32 nr_steps = blocksize / step;
- struct btrfs_csum_ctx csum;
-
- btrfs_csum_init(&csum, fs_info->csum_type);
- for (int i = 0; i < nr_steps; i++) {
- const phys_addr_t paddr = paddrs[i];
+ btrfs_csum_init(&cctx, fs_info->csum_type);
+ while (cur < blocksize) {
+ struct page *page = bio_iter_page(bio, iter);
+ const u32 pg_off = bio_iter_offset(bio, iter);
+ const u32 cur_len = min(bio_iter_len(bio, iter), blocksize - cur);
void *kaddr;
- ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE);
- kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr);
- btrfs_csum_update(&csum, kaddr, step);
+ kaddr = kmap_local_page(page) + pg_off;
+ btrfs_csum_update(&cctx, kaddr, cur_len);
kunmap_local(kaddr);
- }
- btrfs_csum_final(&csum, dest);
-}
-/*
- * Verify the checksum for a single sector without any extra action that depend
- * on the type of I/O.
- *
- * @kaddr must be a properly kmapped address.
- */
-int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum,
- const u8 * const csum_expected)
-{
- btrfs_calculate_block_csum_folio(fs_info, paddr, csum);
- if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0))
- return -EIO;
- return 0;
+ bio_advance_iter_single(bio, &iter, cur_len);
+ cur += cur_len;
+ }
+ btrfs_csum_final(&cctx, csum);
}
/*
@@ -3551,27 +3470,30 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8
* different noncontiguous pages.
*
* @bbio: btrfs_io_bio which contains the csum
- * @dev: device the sector is on
- * @bio_offset: offset to the beginning of the bio (in bytes)
- * @paddrs: physical addresses which back the fs block
+ * @orig_iter: bvec iter pointing to the start of the block
+ * @dev: device the sector is on (optional)
*
* Check if the checksum on a data block is valid. When a checksum mismatch is
* detected, report the error and fill the corrupted range with zero.
*
* Return %true if the sector is ok or had no checksum to start with, else %false.
*/
-bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev,
- u32 bio_offset, const phys_addr_t paddrs[])
+bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio,
+ const struct bvec_iter *orig_iter,
+ struct btrfs_device *dev)
{
struct btrfs_inode *inode = bbio->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ struct bvec_iter iter = *orig_iter;
const u32 blocksize = fs_info->sectorsize;
- const u32 step = min(blocksize, PAGE_SIZE);
- const u32 nr_steps = blocksize / step;
+ const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT;
u64 file_offset = bbio->file_offset + bio_offset;
u64 end = file_offset + blocksize - 1;
u8 *csum_expected;
u8 csum[BTRFS_CSUM_SIZE];
+ u32 cur = 0;
+
+ ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector);
if (!bbio->csum)
return true;
@@ -3587,7 +3509,7 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev,
csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) *
fs_info->csum_size;
- btrfs_calculate_block_csum_pages(fs_info, paddrs, csum);
+ btrfs_csum_one_bio_block(fs_info, &bbio->bio, orig_iter, csum);
if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0))
goto zeroit;
return true;
@@ -3597,8 +3519,16 @@ zeroit:
bbio->mirror_num);
if (dev)
btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS);
- for (int i = 0; i < nr_steps; i++)
- memzero_page(phys_to_page(paddrs[i]), offset_in_page(paddrs[i]), step);
+ while (cur < blocksize) {
+ struct page *page = bio_iter_page(&bbio->bio, iter);
+ const u32 pg_off = bio_iter_offset(&bbio->bio, iter);
+ const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur);
+
+ memzero_page(page, pg_off, cur_len);
+
+ bio_advance_iter_single(&bbio->bio, &iter, cur_len);
+ cur += cur_len;
+ }
return false;
}
@@ -6881,8 +6811,28 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans,
}
} else {
ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name,
- false, BTRFS_I(inode)->dir_index);
- if (unlikely(ret)) {
+ false, BTRFS_I(inode)->dir_index, NULL);
+ if (ret == -ENOMEM) {
+ /*
+ * Orphan the new inode instead of aborting. The inode
+ * item was already written with nlink 1, and discard's
+ * eviction won't delete a bad inode, so nlink 0 must be
+ * persisted here or orphan cleanup would see nlink > 0,
+ * drop the orphan item, and leak the inode.
+ */
+ clear_nlink(inode);
+ /* btrfs_orphan_add() aborts the transaction on failure. */
+ ret = btrfs_orphan_add(trans, BTRFS_I(inode));
+ if (ret)
+ goto discard;
+ ret = btrfs_update_inode(trans, BTRFS_I(inode));
+ if (ret) {
+ btrfs_abort_transaction(trans, ret);
+ goto discard;
+ }
+ ret = -ENOMEM;
+ goto discard;
+ } else if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
goto discard;
}
@@ -6913,7 +6863,8 @@ out:
*/
int btrfs_add_link(struct btrfs_trans_handle *trans,
struct btrfs_inode *parent_inode, struct btrfs_inode *inode,
- const struct fscrypt_str *name, bool add_backref, u64 index)
+ const struct fscrypt_str *name, bool add_backref, u64 index,
+ struct btrfs_dir_index_prealloc *prealloc)
{
int ret = 0;
struct btrfs_key key;
@@ -6939,12 +6890,14 @@ int btrfs_add_link(struct btrfs_trans_handle *trans,
}
/* Nothing to clean up yet */
- if (ret)
+ if (ret) {
+ btrfs_free_delayed_dir_index_prealloc(trans, prealloc);
return ret;
+ }
ret = btrfs_insert_dir_item(trans, name, parent_inode, &key,
- btrfs_inode_type(inode), index);
- if (ret == -EEXIST || ret == -EOVERFLOW)
+ btrfs_inode_type(inode), index, prealloc);
+ if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM)
goto fail_dir_item;
else if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
@@ -7097,7 +7050,7 @@ static int btrfs_link(struct dentry *old_dentry, struct inode *dir,
inode_set_ctime_current(inode);
ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode),
- &fname.disk_name, true, index);
+ &fname.disk_name, true, index, NULL);
if (ret)
goto fail;
@@ -7164,7 +7117,7 @@ static noinline int uncompress_inline(struct btrfs_path *path,
compress_type = btrfs_file_extent_compression(leaf, item);
max_size = btrfs_file_extent_ram_bytes(leaf, item);
inline_size = btrfs_file_extent_inline_item_len(leaf, path->slots[0]);
- tmp = kmalloc(inline_size, GFP_NOFS);
+ tmp = kvmalloc(inline_size, GFP_NOFS);
if (!tmp)
return -ENOMEM;
ptr = btrfs_file_extent_inline_start(item);
@@ -7185,7 +7138,7 @@ static noinline int uncompress_inline(struct btrfs_path *path,
if (max_size < blocksize)
folio_zero_range(folio, max_size, blocksize - max_size);
- kfree(tmp);
+ kvfree(tmp);
return ret;
}
@@ -7281,16 +7234,6 @@ struct extent_map *btrfs_get_extent(struct btrfs_inode *inode,
/* Chances are we'll be called again, so go ahead and do readahead */
path->reada = READA_FORWARD;
- /*
- * The same explanation in load_free_space_cache applies here as well,
- * we only read when we're loading the free space cache, and at that
- * point the commit_root has everything we need.
- */
- if (btrfs_is_free_space_inode(inode)) {
- path->search_commit_root = true;
- path->skip_locking = true;
- }
-
ret = btrfs_lookup_file_extent(NULL, root, path, objectid, start, 0);
if (ret < 0) {
goto out;
@@ -8147,7 +8090,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode)
struct btrfs_ordered_extent *ordered;
struct btrfs_inode *inode = BTRFS_I(vfs_inode);
struct btrfs_root *root = inode->root;
- bool freespace_inode;
WARN_ON(!hlist_empty(&vfs_inode->i_dentry));
WARN_ON(vfs_inode->i_data.nrpages);
@@ -8170,12 +8112,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode)
if (!root)
return;
- /*
- * If this is a free space inode do not take the ordered extents lockdep
- * map.
- */
- freespace_inode = btrfs_is_free_space_inode(inode);
-
while (1) {
ordered = btrfs_lookup_first_ordered_extent(inode, (u64)-1);
if (!ordered)
@@ -8185,8 +8121,7 @@ void btrfs_destroy_inode(struct inode *vfs_inode)
"found ordered extent %llu %llu on inode cleanup",
ordered->file_offset, ordered->num_bytes);
- if (!freespace_inode)
- btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent);
+ btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent);
btrfs_remove_ordered_extent(ordered);
btrfs_put_ordered_extent(ordered);
@@ -8511,14 +8446,14 @@ static int btrfs_rename_exchange(struct inode *old_dir,
}
ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode),
- new_name, false, old_idx);
+ new_name, false, old_idx, NULL);
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
goto out_fail;
}
ret = btrfs_add_link(trans, BTRFS_I(old_dir), BTRFS_I(new_inode),
- old_name, false, new_idx);
+ old_name, false, new_idx, NULL);
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
goto out_fail;
@@ -8591,6 +8526,7 @@ static int btrfs_rename(struct mnt_idmap *idmap,
struct inode *new_inode = d_inode(new_dentry);
struct inode *old_inode = d_inode(old_dentry);
struct btrfs_rename_ctx rename_ctx;
+ struct btrfs_dir_index_prealloc *prealloc = NULL;
u64 index = 0;
int ret;
int ret2;
@@ -8714,6 +8650,23 @@ static int btrfs_rename(struct mnt_idmap *idmap,
if (ret)
goto out_fail;
+ /*
+ * When not overwriting an existing entry, pre-allocate the delayed dir
+ * index now so that ENOMEM is returned before any btree modifications.
+ * For the overwrite case, too many btree changes have already happened
+ * by the time btrfs_add_link() is called.
+ */
+ if (!new_inode) {
+ prealloc = btrfs_prealloc_delayed_dir_index(BTRFS_I(new_dir),
+ new_fname.disk_name.name,
+ new_fname.disk_name.len);
+ if (IS_ERR(prealloc)) {
+ ret = PTR_ERR(prealloc);
+ prealloc = NULL;
+ goto out_fail;
+ }
+ }
+
BTRFS_I(old_inode)->dir_index = 0ULL;
if (unlikely(old_ino == BTRFS_FIRST_FREE_OBJECTID)) {
/* force full log commit if subvolume involved. */
@@ -8809,7 +8762,8 @@ static int btrfs_rename(struct mnt_idmap *idmap,
}
ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode),
- &new_fname.disk_name, false, index);
+ &new_fname.disk_name, false, index, prealloc);
+ prealloc = NULL;
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
goto out_fail;
@@ -8834,6 +8788,7 @@ static int btrfs_rename(struct mnt_idmap *idmap,
}
}
out_fail:
+ btrfs_free_delayed_dir_index_prealloc(trans, prealloc);
if (logs_pinned) {
btrfs_end_log_trans(root);
btrfs_end_log_trans(dest);
@@ -9148,14 +9103,13 @@ out_inode:
}
static struct btrfs_trans_handle *insert_prealloc_file_extent(
- struct btrfs_trans_handle *trans_in,
struct btrfs_inode *inode,
struct btrfs_key *ins,
u64 file_offset)
{
struct btrfs_file_extent_item stack_fi;
struct btrfs_replace_extent_info extent_info;
- struct btrfs_trans_handle *trans = trans_in;
+ struct btrfs_trans_handle *trans;
struct btrfs_path *path;
u64 start = ins->objectid;
u64 len = ins->offset;
@@ -9176,15 +9130,6 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent(
if (ret < 0)
return ERR_PTR(ret);
- if (trans) {
- ret = insert_reserved_file_extent(trans, inode,
- file_offset, &stack_fi,
- true, qgroup_released);
- if (ret)
- goto free_qgroup;
- return trans;
- }
-
extent_info.disk_offset = start;
extent_info.disk_len = len;
extent_info.data_offset = 0;
@@ -9224,12 +9169,12 @@ free_qgroup:
return ERR_PTR(ret);
}
-static int __btrfs_prealloc_file_range(struct inode *inode, int mode,
- u64 start, u64 num_bytes, u64 min_size,
- loff_t actual_len, u64 *alloc_hint,
- struct btrfs_trans_handle *trans)
+int btrfs_prealloc_file_range(struct inode *inode, int mode,
+ u64 start, u64 num_bytes, u64 min_size,
+ loff_t actual_len, u64 *alloc_hint)
{
struct btrfs_fs_info *fs_info = inode_to_fs_info(inode);
+ struct btrfs_trans_handle *trans;
struct extent_map *em;
struct btrfs_root *root = BTRFS_I(inode)->root;
struct btrfs_key ins;
@@ -9239,11 +9184,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode,
u64 cur_bytes;
u64 last_alloc = (u64)-1;
int ret = 0;
- bool own_trans = true;
u64 end = start + num_bytes - 1;
- if (trans)
- own_trans = false;
while (num_bytes > 0) {
cur_bytes = min_t(u64, num_bytes, SZ_256M);
cur_bytes = max(cur_bytes, min_size);
@@ -9269,8 +9211,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode,
clear_offset += ins.offset;
last_alloc = ins.offset;
- trans = insert_prealloc_file_extent(trans, BTRFS_I(inode),
- &ins, cur_offset);
+ trans = insert_prealloc_file_extent(BTRFS_I(inode), &ins,
+ cur_offset);
/*
* Now that we inserted the prealloc extent we can finally
* decrement the number of reservations in the block group.
@@ -9342,8 +9284,7 @@ next:
range_start, range_end - range_start);
if (ret) {
btrfs_abort_transaction(trans, ret);
- if (own_trans)
- btrfs_end_transaction(trans);
+ btrfs_end_transaction(trans);
break;
}
@@ -9355,15 +9296,11 @@ next:
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
- if (own_trans)
- btrfs_end_transaction(trans);
+ btrfs_end_transaction(trans);
break;
}
- if (own_trans) {
- btrfs_end_transaction(trans);
- trans = NULL;
- }
+ btrfs_end_transaction(trans);
}
if (clear_offset < end)
btrfs_free_reserved_data_space(BTRFS_I(inode), NULL, clear_offset,
@@ -9371,24 +9308,6 @@ next:
return ret;
}
-int btrfs_prealloc_file_range(struct inode *inode, int mode,
- u64 start, u64 num_bytes, u64 min_size,
- loff_t actual_len, u64 *alloc_hint)
-{
- return __btrfs_prealloc_file_range(inode, mode, start, num_bytes,
- min_size, actual_len, alloc_hint,
- NULL);
-}
-
-int btrfs_prealloc_file_range_trans(struct inode *inode,
- struct btrfs_trans_handle *trans, int mode,
- u64 start, u64 num_bytes, u64 min_size,
- loff_t actual_len, u64 *alloc_hint)
-{
- return __btrfs_prealloc_file_range(inode, mode, start, num_bytes,
- min_size, actual_len, alloc_hint, trans);
-}
-
/*
* NOTE: in case you are adding MAY_EXEC check for directories:
* we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to
@@ -9623,7 +9542,6 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode,
struct completion sync_reads;
unsigned long i = 0;
struct btrfs_bio *bbio;
- int ret;
/*
* Fast path for synchronous reads which completes in this call, io_uring
@@ -9670,10 +9588,10 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode,
if (uring_ctx) {
if (refcount_dec_and_test(&priv->pending_refs)) {
- ret = blk_status_to_errno(READ_ONCE(priv->status));
- btrfs_uring_read_extent_endio(uring_ctx, ret);
+ int err = blk_status_to_errno(READ_ONCE(priv->status));
+
+ btrfs_uring_read_extent_endio(uring_ctx, err);
kfree(priv);
- return ret;
}
return -EIOCBQUEUED;
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
index e4b2da31a0d5..52aab510aea0 100644
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -3881,7 +3881,7 @@ static long btrfs_ioctl_quota_rescan_status(struct btrfs_fs_info *fs_info,
if (!capable(CAP_SYS_ADMIN))
return -EPERM;
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) {
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) {
qsa.flags = 1;
qsa.progress = fs_info->qgroup_rescan_progress.objectid;
}
@@ -4601,7 +4601,7 @@ static void btrfs_uring_read_finished(struct io_tw_req tw_req, io_tw_token_t tw)
size_t page_offset;
ssize_t ret;
- /* The inode lock has already been acquired in btrfs_uring_read_extent. */
+ /* The inode lock has already been acquired in btrfs_encoded_read(). */
btrfs_lockdep_inode_acquire(inode, i_rwsem);
if (priv->err) {
@@ -4667,7 +4667,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter,
struct iovec *iov, struct io_uring_cmd *cmd)
{
struct btrfs_inode *inode = BTRFS_I(file_inode(iocb->ki_filp));
- struct extent_io_tree *io_tree = &inode->io_tree;
struct page **pages = NULL;
struct btrfs_uring_priv *priv = NULL;
unsigned long nr_pages;
@@ -4723,8 +4722,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter,
return -EIOCBQUEUED;
out_fail:
- btrfs_unlock_extent(io_tree, start, lockend, &cached_state);
- btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED);
kfree(priv);
for (int i = 0; i < nr_pages; i++) {
if (pages[i])
@@ -4752,9 +4749,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue
struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd);
struct btrfs_uring_encoded_data *data = NULL;
- if (cmd->flags & IORING_URING_CMD_REISSUE)
- data = bc->data;
-
if (!capable(CAP_SYS_ADMIN)) {
ret = -EPERM;
goto out_acct;
@@ -4836,8 +4830,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue
ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state,
&disk_bytenr, &disk_io_size);
- if (ret == -EAGAIN)
- goto out_acct;
if (ret < 0 && ret != -EIOCBQUEUED)
goto out_free;
@@ -4865,8 +4857,10 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue
cached_state, disk_bytenr, disk_io_size,
count, data->args.compression,
data->iov, cmd);
-
- goto out_acct;
+ if (ret == -EIOCBQUEUED)
+ goto out_acct;
+ btrfs_unlock_extent(io_tree, start, lockend, &cached_state);
+ btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED);
}
out_free:
@@ -4877,8 +4871,10 @@ out_acct:
add_rchar(current, ret);
inc_syscr(current);
- if (ret != -EIOCBQUEUED && ret != -EAGAIN)
+ if (ret != -EIOCBQUEUED) {
kfree(data);
+ bc->data = NULL;
+ }
return ret;
}
@@ -4890,12 +4886,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu
struct kiocb kiocb;
ssize_t ret;
void __user *sqe_addr;
- struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd);
struct btrfs_uring_encoded_data *data = NULL;
- if (cmd->flags & IORING_URING_CMD_REISSUE)
- data = bc->data;
-
if (!capable(CAP_SYS_ADMIN)) {
ret = -EPERM;
goto out_acct;
@@ -4907,6 +4899,11 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu
goto out_acct;
}
+ if (issue_flags & IO_URING_F_NONBLOCK) {
+ ret = -EAGAIN;
+ goto out_acct;
+ }
+
if (!data) {
data = kzalloc_obj(*data, GFP_NOFS);
if (!data) {
@@ -4914,8 +4911,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu
goto out_acct;
}
- bc->data = data;
-
if (issue_flags & IO_URING_F_COMPAT) {
#if defined(CONFIG_64BIT) && defined(CONFIG_COMPAT)
struct btrfs_ioctl_encoded_io_args_32 args32;
@@ -4975,11 +4970,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu
}
}
- if (issue_flags & IO_URING_F_NONBLOCK) {
- ret = -EAGAIN;
- goto out_acct;
- }
-
pos = data->args.offset;
ret = rw_verify_area(WRITE, file, &pos, data->args.len);
if (ret < 0)
@@ -5005,8 +4995,7 @@ out_acct:
add_wchar(current, ret);
inc_syscw(current);
- if (ret != -EAGAIN)
- kfree(data);
+ kfree(data);
return ret;
}
diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c
index b32d4eabe0ab..df74c75d6c29 100644
--- a/fs/btrfs/ordered-data.c
+++ b/fs/btrfs/ordered-data.c
@@ -417,13 +417,10 @@ static bool can_finish_ordered_extent(struct btrfs_ordered_extent *ordered,
static void btrfs_queue_ordered_fn(struct btrfs_ordered_extent *ordered)
{
- struct btrfs_inode *inode = ordered->inode;
- struct btrfs_fs_info *fs_info = inode->root->fs_info;
- struct btrfs_workqueue *wq = btrfs_is_free_space_inode(inode) ?
- fs_info->endio_freespace_worker : fs_info->endio_write_workers;
+ struct btrfs_fs_info *fs_info = ordered->inode->root->fs_info;
btrfs_init_work(&ordered->work, finish_ordered_fn, NULL);
- btrfs_queue_work(wq, &ordered->work);
+ btrfs_queue_work(fs_info->endio_write_workers, &ordered->work);
}
void btrfs_finish_ordered_extent(struct btrfs_ordered_extent *ordered,
@@ -657,13 +654,6 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry)
struct btrfs_fs_info *fs_info = root->fs_info;
struct rb_node *node;
bool pending;
- bool freespace_inode;
-
- /*
- * If this is a free space inode the thread has not acquired the ordered
- * extents lockdep map.
- */
- freespace_inode = btrfs_is_free_space_inode(btrfs_inode);
btrfs_lockdep_acquire(fs_info, btrfs_trans_pending_ordered);
/* This is paired with alloc_ordered_extent(). */
@@ -738,8 +728,7 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry)
}
spin_unlock(&root->ordered_extent_lock);
wake_up(&entry->wait);
- if (!freespace_inode)
- btrfs_lockdep_release(fs_info, btrfs_ordered_extent);
+ btrfs_lockdep_release(fs_info, btrfs_ordered_extent);
}
static void btrfs_run_ordered_extent_work(struct btrfs_work *work)
@@ -870,17 +859,10 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry,
u64 start = entry->file_offset;
u64 end = start + entry->num_bytes - 1;
struct btrfs_inode *inode = entry->inode;
- bool freespace_inode;
trace_btrfs_ordered_extent_start(inode, entry);
/*
- * If this is a free space inode do not take the ordered extents lockdep
- * map.
- */
- freespace_inode = btrfs_is_free_space_inode(inode);
-
- /*
* pages in the range can be dirty, clean or writeback. We
* start IO on any dirty ones so the wait doesn't stall waiting
* for the flusher thread to find them
@@ -899,8 +881,7 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry,
}
}
- if (!freespace_inode)
- btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent);
+ btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent);
wait_event(entry->wait, test_bit(BTRFS_ORDERED_COMPLETE, &entry->flags));
}
diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c
index f68b696b4bf7..05e35eb126dc 100644
--- a/fs/btrfs/qgroup.c
+++ b/fs/btrfs/qgroup.c
@@ -34,7 +34,7 @@ enum btrfs_qgroup_mode btrfs_qgroup_mode(const struct btrfs_fs_info *fs_info)
{
if (!test_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags))
return BTRFS_QGROUP_MODE_DISABLED;
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE)
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags))
return BTRFS_QGROUP_MODE_SIMPLE;
return BTRFS_QGROUP_MODE_FULL;
}
@@ -384,14 +384,14 @@ static bool squota_check_parent_usage(struct btrfs_fs_info *fs_info, struct btrf
__printf(2, 3)
static void qgroup_mark_inconsistent(struct btrfs_fs_info *fs_info, const char *fmt, ...)
{
- const u64 old_flags = fs_info->qgroup_flags;
+ const unsigned long old_flags = fs_info->qgroup_flags;
if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE)
return;
- fs_info->qgroup_flags |= (BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT |
- BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN |
- BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING);
- if (!(old_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) {
+ set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
+ set_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags);
+ set_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags);
+ if (!test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &old_flags)) {
struct va_format vaf;
va_list args;
@@ -426,7 +426,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info)
struct extent_buffer *l;
int slot;
int ret = 0;
- u64 flags = 0;
u64 rescan_progress = 0;
if (!fs_info->quota_root)
@@ -473,8 +472,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info)
"old qgroup version, quota disabled");
goto out;
}
- fs_info->qgroup_flags = btrfs_qgroup_status_flags(l, ptr);
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE)
+ if (btrfs_qgroup_status_flags(l, ptr) > ULONG_MAX) {
+ btrfs_err(fs_info, "invalid qgroup status flags, quota disabled");
+ goto out;
+ }
+ fs_info->qgroup_flags = (unsigned long)btrfs_qgroup_status_flags(l, ptr);
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags))
qgroup_read_enable_gen(fs_info, l, slot, ptr);
else if (btrfs_qgroup_status_generation(l, ptr) != fs_info->generation)
qgroup_mark_inconsistent(fs_info, "qgroup generation mismatch");
@@ -609,14 +612,13 @@ next2:
}
out:
btrfs_free_path(path);
- fs_info->qgroup_flags |= flags;
if (ret >= 0) {
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON)
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))
set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags);
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN)
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))
ret = qgroup_rescan_init(fs_info, rescan_progress, 0);
} else {
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN;
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags);
btrfs_sysfs_del_qgroups(fs_info);
}
@@ -1101,9 +1103,9 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info,
struct btrfs_qgroup_status_item);
btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid);
btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION);
- fs_info->qgroup_flags = BTRFS_QGROUP_STATUS_FLAG_ON;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags);
if (simple) {
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags);
btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA);
/*
* Set the enable generation to the next transaction, as we cannot
@@ -1113,7 +1115,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info,
*/
btrfs_set_qgroup_status_enable_gen(leaf, ptr, trans->transid + 1);
} else {
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
}
btrfs_set_qgroup_status_flags(leaf, ptr, fs_info->qgroup_flags &
BTRFS_QGROUP_STATUS_FLAGS_MASK);
@@ -1403,8 +1405,14 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info)
spin_lock(&fs_info->qgroup_lock);
quota_root = fs_info->quota_root;
fs_info->quota_root = NULL;
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON;
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE;
+ /*
+ * Clear all on-disk and runtime bits, except RESCAN related ones, that
+ * are either handled by rescan thread, or the caller who rejects rescan.
+ */
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags);
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags);
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
+ clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags);
fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT;
spin_unlock(&fs_info->qgroup_lock);
@@ -1554,7 +1562,7 @@ static int quick_update_accounting(struct btrfs_fs_info *fs_info,
}
out:
if (ret)
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
return ret;
}
@@ -1875,7 +1883,7 @@ int btrfs_remove_qgroup(struct btrfs_trans_handle *trans, u64 qgroupid)
* very frequently.
*/
if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_FULL &&
- !(fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) {
+ !test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags)) {
if (unlikely(qgroup->rfer || qgroup->excl ||
qgroup->rfer_cmpr || qgroup->excl_cmpr)) {
DEBUG_WARN();
@@ -2120,7 +2128,7 @@ int btrfs_qgroup_trace_extent_post(struct btrfs_trans_handle *trans,
*/
ASSERT(trans != NULL);
- if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)
+ if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags))
return 0;
ret = btrfs_find_all_roots(&ctx, true);
@@ -2740,6 +2748,24 @@ walk_down:
return 0;
}
+void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level)
+{
+ u8 drop_subtree_thres;
+
+ if (btrfs_qgroup_mode(fs_info) != BTRFS_QGROUP_MODE_FULL)
+ return;
+
+ if (!btrfs_is_fstree(rootid))
+ return;
+
+ spin_lock(&fs_info->qgroup_lock);
+ drop_subtree_thres = fs_info->qgroup_drop_subtree_thres;
+ spin_unlock(&fs_info->qgroup_lock);
+
+ if (level >= drop_subtree_thres)
+ qgroup_mark_inconsistent(fs_info, "subtree level reached threshold");
+}
+
static void qgroup_iterator_nested_add(struct list_head *head, struct btrfs_qgroup *qgroup)
{
if (!list_empty(&qgroup->nested_iterator))
@@ -2961,7 +2987,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr,
* we can't just exit here.
*/
if (!btrfs_qgroup_full_accounting(fs_info) ||
- fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)
+ test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags))
goto out_free;
if (new_roots) {
@@ -2983,7 +3009,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr,
num_bytes, nr_old_roots, nr_new_roots);
mutex_lock(&fs_info->qgroup_rescan_lock);
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) {
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) {
if (fs_info->qgroup_rescan_progress.objectid <= bytenr) {
mutex_unlock(&fs_info->qgroup_rescan_lock);
ret = 0;
@@ -3044,8 +3070,8 @@ int btrfs_qgroup_account_extents(struct btrfs_trans_handle *trans)
num_dirty_extents++;
trace_btrfs_qgroup_account_extents(fs_info, record, bytenr);
- if (!ret && !(fs_info->qgroup_flags &
- BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)) {
+ if (!ret && !test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING,
+ &fs_info->qgroup_flags)) {
struct btrfs_backref_walk_ctx ctx = { 0 };
ctx.bytenr = bytenr;
@@ -3152,9 +3178,9 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans)
spin_lock(&fs_info->qgroup_lock);
}
if (btrfs_qgroup_enabled(fs_info))
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_ON;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags);
else
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON;
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags);
spin_unlock(&fs_info->qgroup_lock);
ret = update_qgroup_status_item(trans);
@@ -3844,7 +3870,7 @@ static bool rescan_should_stop(struct btrfs_fs_info *fs_info)
return true;
if (!btrfs_qgroup_enabled(fs_info))
return true;
- if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN)
+ if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags))
return true;
return false;
}
@@ -3894,12 +3920,10 @@ out:
btrfs_free_path(path);
mutex_lock(&fs_info->qgroup_rescan_lock);
- if (ret > 0 &&
- fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT) {
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT;
- } else if (ret < 0 || stopped) {
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT;
- }
+ if (ret > 0)
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
+ else if (ret < 0 || stopped)
+ set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
mutex_unlock(&fs_info->qgroup_rescan_lock);
/*
@@ -3923,9 +3947,9 @@ out:
}
mutex_lock(&fs_info->qgroup_rescan_lock);
- if (!stopped ||
- fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN)
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN;
+ if (!stopped || test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN,
+ &fs_info->qgroup_flags))
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags);
if (trans) {
int ret2 = update_qgroup_status_item(trans);
@@ -3935,7 +3959,7 @@ out:
}
}
fs_info->qgroup_rescan_running = false;
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN;
+ clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags);
complete_all(&fs_info->qgroup_rescan_completion);
mutex_unlock(&fs_info->qgroup_rescan_lock);
@@ -3946,7 +3970,7 @@ out:
if (stopped) {
btrfs_info(fs_info, "qgroup scan paused");
- } else if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) {
+ } else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) {
btrfs_info(fs_info, "qgroup scan cancelled");
} else if (ret >= 0) {
btrfs_info(fs_info, "qgroup scan completed%s",
@@ -3973,13 +3997,11 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid,
if (!init_flags) {
/* we're resuming qgroup rescan at mount time */
- if (!(fs_info->qgroup_flags &
- BTRFS_QGROUP_STATUS_FLAG_RESCAN)) {
+ if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))) {
btrfs_debug(fs_info,
"qgroup rescan init failed, qgroup rescan is not queued");
ret = -EINVAL;
- } else if (!(fs_info->qgroup_flags &
- BTRFS_QGROUP_STATUS_FLAG_ON)) {
+ } else if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))) {
btrfs_debug(fs_info,
"qgroup rescan init failed, qgroup is not enabled");
ret = -ENOTCONN;
@@ -3992,10 +4014,12 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid,
mutex_lock(&fs_info->qgroup_rescan_lock);
if (init_flags) {
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) {
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN,
+ &fs_info->qgroup_flags) ||
+ test_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN,
+ &fs_info->qgroup_flags)) {
ret = -EINPROGRESS;
- } else if (!(fs_info->qgroup_flags &
- BTRFS_QGROUP_STATUS_FLAG_ON)) {
+ } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) {
btrfs_debug(fs_info,
"qgroup rescan init failed, qgroup is not enabled");
ret = -ENOTCONN;
@@ -4008,13 +4032,13 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid,
mutex_unlock(&fs_info->qgroup_rescan_lock);
return ret;
}
- fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_RESCAN;
+ set_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags);
}
memset(&fs_info->qgroup_rescan_progress, 0,
sizeof(fs_info->qgroup_rescan_progress));
- fs_info->qgroup_flags &= ~(BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN |
- BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING);
+ clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags);
+ clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags);
fs_info->qgroup_rescan_progress.objectid = progress_objectid;
init_completion(&fs_info->qgroup_rescan_completion);
mutex_unlock(&fs_info->qgroup_rescan_lock);
@@ -4065,7 +4089,7 @@ btrfs_qgroup_rescan(struct btrfs_fs_info *fs_info)
ret = btrfs_commit_current_transaction(fs_info->fs_root);
if (ret) {
- fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN;
+ clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags);
return ret;
}
@@ -4118,7 +4142,7 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info,
void
btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info)
{
- if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) {
+ if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) {
mutex_lock(&fs_info->qgroup_rescan_lock);
fs_info->qgroup_rescan_running = true;
btrfs_queue_work(fs_info->qgroup_rescan_workers,
diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h
index 80dd2dacd56d..c64b26b09c22 100644
--- a/fs/btrfs/qgroup.h
+++ b/fs/btrfs/qgroup.h
@@ -121,8 +121,19 @@ struct btrfs_qgroup_swapped_blocks;
* To minimize the chance of collision with new persisted status flags, these
* count backwards from the MSB.
*/
-#define BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN (1ULL << 63)
-#define BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING (1ULL << 62)
+#define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1)
+#define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2)
+
+/*
+ * No new rescan allowed when set.
+ *
+ * During huge subtree dropping, qgroup will be marked inconsistent, and skip
+ * all future accounting to avoid long stall. But, an immediate rescan will
+ * re-enable qgroup and still stall the system.
+ *
+ * This bit is to avoid such rescan during the duration of a subvolume dropping.
+ */
+#define BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN (BITS_PER_LONG - 3)
#define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3)
@@ -365,6 +376,7 @@ int btrfs_qgroup_trace_leaf_items(struct btrfs_trans_handle *trans,
int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans,
struct extent_buffer *root_eb,
u64 root_gen, int root_level);
+void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level);
int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr,
u64 num_bytes, struct ulist *old_roots,
struct ulist *new_roots);
diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c
index 1ee52a9dcee3..8ec24dbb180f 100644
--- a/fs/btrfs/raid56.c
+++ b/fs/btrfs/raid56.c
@@ -953,7 +953,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status)
/*
* Clear the data bitmap, as the rbio may be cached for later usage.
- * do this before before unlock_stripe() so there will be no new bio
+ * do this before unlock_stripe() so there will be no new bio
* for this bio.
*/
bitmap_clear(&rbio->dbitmap, 0, rbio->stripe_nsectors);
@@ -988,7 +988,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status)
* as possible, and only use stripe_sectors as fallback.
*
* Return NULL if bio_list_only is set but the specified sector has no
- * coresponding bio.
+ * corresponding bio.
*/
static phys_addr_t *sector_paddrs_in_rbio(struct btrfs_raid_bio *rbio,
int stripe_nr, int sector_nr,
@@ -1450,10 +1450,7 @@ static int rmw_assemble_write_bios(struct btrfs_raid_bio *rbio,
/* We should have at least one data sector. */
ASSERT(bitmap_weight(&rbio->dbitmap, rbio->stripe_nsectors));
- /*
- * Reset errors, as we may have errors inherited from from degraded
- * write.
- */
+ /* Reset errors, as we may have errors inherited from degraded write. */
bitmap_clear(rbio->error_bitmap, 0, rbio->nr_sectors);
/*
@@ -1652,12 +1649,7 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio,
struct bio *bio)
{
struct btrfs_fs_info *fs_info = rbio->bioc->fs_info;
- const u32 step = min(fs_info->sectorsize, PAGE_SIZE);
- const u32 nr_steps = rbio->sector_nsteps;
int total_sector_nr = get_bio_sector_nr(rbio, bio);
- u32 offset = 0;
- phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE];
- phys_addr_t paddr;
/* No data csum for the whole stripe, no need to verify. */
if (!rbio->csum_bitmap || !rbio->csum_buf)
@@ -1667,28 +1659,20 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio,
if (total_sector_nr >= rbio->nr_data * rbio->stripe_nsectors)
return;
- btrfs_bio_for_each_block_all(paddr, bio, step) {
+ for (struct bvec_iter iter = init_bvec_iter_for_bio(bio);
+ iter.bi_size;
+ bio_advance_iter(bio, &iter, fs_info->sectorsize), total_sector_nr++) {
u8 csum_buf[BTRFS_CSUM_SIZE];
u8 *expected_csum;
- paddrs[(offset / step) % nr_steps] = paddr;
- offset += step;
-
- /* Not yet covering the full fs block, continue to the next step. */
- if (!IS_ALIGNED(offset, fs_info->sectorsize))
- continue;
-
/* No csum for this sector, skip to the next sector. */
- if (!test_bit(total_sector_nr, rbio->csum_bitmap)) {
- total_sector_nr++;
+ if (!test_bit(total_sector_nr, rbio->csum_bitmap))
continue;
- }
expected_csum = rbio->csum_buf + total_sector_nr * fs_info->csum_size;
- btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf);
+ btrfs_csum_one_bio_block(fs_info, bio, &iter, csum_buf);
if (unlikely(memcmp(csum_buf, expected_csum, fs_info->csum_size) != 0))
set_bit(total_sector_nr, rbio->error_bitmap);
- total_sector_nr++;
}
}
@@ -1879,6 +1863,27 @@ void raid56_parity_write(struct bio *bio, struct btrfs_io_context *bioc)
start_async_work(rbio, rmw_rbio_work);
}
+static void calculate_block_csum_paddrs(struct btrfs_fs_info *fs_info,
+ const phys_addr_t paddrs[], u8 *dest)
+{
+ const u32 blocksize = fs_info->sectorsize;
+ const u32 step = min(blocksize, PAGE_SIZE);
+ const u32 nr_steps = blocksize / step;
+ struct btrfs_csum_ctx csum;
+
+ btrfs_csum_init(&csum, fs_info->csum_type);
+ for (int i = 0; i < nr_steps; i++) {
+ const phys_addr_t paddr = paddrs[i];
+ void *kaddr;
+
+ ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE);
+ kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr);
+ btrfs_csum_update(&csum, kaddr, step);
+ kunmap_local(kaddr);
+ }
+ btrfs_csum_final(&csum, dest);
+}
+
static int verify_one_sector(struct btrfs_raid_bio *rbio,
int stripe_nr, int sector_nr)
{
@@ -1906,7 +1911,7 @@ static int verify_one_sector(struct btrfs_raid_bio *rbio,
csum_expected = rbio->csum_buf +
(stripe_nr * rbio->stripe_nsectors + sector_nr) *
fs_info->csum_size;
- btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf);
+ calculate_block_csum_paddrs(fs_info, paddrs, csum_buf);
if (unlikely(memcmp(csum_buf, csum_expected, fs_info->csum_size) != 0))
return -EIO;
return 0;
@@ -2624,7 +2629,7 @@ static int alloc_rbio_essential_pages(struct btrfs_raid_bio *rbio)
return 0;
}
-/* Return true if the content of the step matches the caclulated one. */
+/* Return true if the content of the step matches the calculated one. */
static bool verify_one_parity_step(struct btrfs_raid_bio *rbio,
void *pointers[], unsigned int sector_nr,
unsigned int step_nr)
diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c
index da54db75e7a9..630a7ad8f8e1 100644
--- a/fs/btrfs/relocation.c
+++ b/fs/btrfs/relocation.c
@@ -3357,7 +3357,7 @@ truncate:
goto out;
}
- ret = btrfs_truncate_free_space_cache(trans, block_group, inode);
+ ret = btrfs_truncate_free_space_cache(trans, inode);
btrfs_end_transaction(trans);
btrfs_btree_balance_dirty(fs_info);
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
index 5c59b9abedcd..c523bf950c89 100644
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -7023,7 +7023,7 @@ static int changed_extent(struct send_ctx *sctx,
* get modified or replaced with a new one). Note that deduplication
* updates the inode item, but it only changes the iversion (sequence
* field in the inode item) of the inode, so if a file is deduplicated
- * the same amount of times in both the parent and send snapshots, its
+ * the same number of times in both the parent and send snapshots, its
* iversion becomes the same in both snapshots, whence the inode item is
* the same on both snapshots.
*/
diff --git a/fs/btrfs/space-info.c b/fs/btrfs/space-info.c
index 39a28e1bec8a..01018152c054 100644
--- a/fs/btrfs/space-info.c
+++ b/fs/btrfs/space-info.c
@@ -1704,7 +1704,6 @@ static int handle_reserve_ticket(struct btrfs_space_info *space_info,
evict_flush_states,
ARRAY_SIZE(evict_flush_states));
break;
- case BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE:
case BTRFS_RESERVE_FLUSH_ZONED_RELOCATION:
priority_reclaim_data_space(space_info, ticket);
break;
@@ -1968,7 +1967,6 @@ int btrfs_reserve_data_bytes(struct btrfs_space_info *space_info, u64 bytes,
int ret;
ASSERT(flush == BTRFS_RESERVE_FLUSH_DATA ||
- flush == BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE ||
flush == BTRFS_RESERVE_FLUSH_ZONED_RELOCATION ||
flush == BTRFS_RESERVE_NO_FLUSH, "flush=%d", flush);
ASSERT(!current->journal_info || flush != BTRFS_RESERVE_FLUSH_DATA,
diff --git a/fs/btrfs/space-info.h b/fs/btrfs/space-info.h
index aa836e8a9d4a..d0130c8ba3dd 100644
--- a/fs/btrfs/space-info.h
+++ b/fs/btrfs/space-info.h
@@ -66,7 +66,6 @@ enum btrfs_reserve_flush_enum {
* Can be interrupted by a fatal signal.
*/
BTRFS_RESERVE_FLUSH_DATA,
- BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE,
BTRFS_RESERVE_FLUSH_ALL,
/*
@@ -82,9 +81,6 @@ enum btrfs_reserve_flush_enum {
* priority flushing for this, because otherwise we can deadlock on
* waiting for a ticket, that cannot be granted, because we cannot do
* any allocations.
- *
- * Apart from being specific to zoned relocation, it is equal to
- * BTRFS_FLUSH_FREE_SPACE_INODE.
*/
BTRFS_RESERVE_FLUSH_ZONED_RELOCATION,
diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c
index ddb620ac241b..14ed0ed823a3 100644
--- a/fs/btrfs/super.c
+++ b/fs/btrfs/super.c
@@ -515,7 +515,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
btrfs_warn(NULL,
"v1 space cache is deprecated, falling back to no space cache");
btrfs_set_opt(ctx->mount_opt, NOSPACECACHE);
- btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
break;
case Opt_space_cache_version:
@@ -524,11 +523,9 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
btrfs_warn(NULL,
"v1 space cache is deprecated, falling back to no space cache");
btrfs_set_opt(ctx->mount_opt, NOSPACECACHE);
- btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
break;
case Opt_space_cache_v2:
- btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
btrfs_set_opt(ctx->mount_opt, FREE_SPACE_TREE);
break;
default:
@@ -705,13 +702,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info,
if (btrfs_check_mountopts_zoned(info, mount_opt))
ret = false;
- if (!test_bit(BTRFS_FS_STATE_REMOUNTING, &info->fs_state)) {
- if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) {
- btrfs_warn(info,
-"space cache v1 is being deprecated and will be removed in a future release, please use -o space_cache=v2");
- }
- }
-
return ret;
}
@@ -729,14 +719,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info,
*/
void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info)
{
- if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) {
- btrfs_info(fs_info,
- "forcing free space tree for sector size %u with page size %lu",
- fs_info->sectorsize, PAGE_SIZE);
- btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE);
- btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE);
- }
-
/*
* At this point our mount options are populated, so we only mess with
* these settings if we don't have any settings already.
@@ -751,20 +733,17 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info)
return;
}
- if (btrfs_test_opt(fs_info, SPACE_CACHE))
- return;
-
if (btrfs_test_opt(fs_info, NOSPACECACHE))
return;
/*
* At this point we don't have explicit options set by the user, set
- * them ourselves based on the state of the file system.
+ * them ourselves based on the state of the file system. An existing
+ * v1 space cache is no longer used and gets cleaned up once the
+ * filesystem is mounted read-write.
*/
if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE))
btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE);
- else if (btrfs_free_space_cache_v1_active(fs_info))
- btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE);
}
static void set_device_specific_options(struct btrfs_fs_info *fs_info)
@@ -1107,9 +1086,7 @@ static int btrfs_show_options(struct seq_file *seq, struct dentry *dentry)
seq_puts(seq, ",discard=async");
if (!(info->sb->s_flags & SB_POSIXACL))
seq_puts(seq, ",noacl");
- if (btrfs_free_space_cache_v1_active(info))
- seq_puts(seq, ",space_cache");
- else if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE))
+ if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE))
seq_puts(seq, ",space_cache=v2");
else
seq_puts(seq, ",nospace_cache");
@@ -1243,7 +1220,6 @@ static void btrfs_resize_thread_pool(struct btrfs_fs_info *fs_info,
workqueue_set_max_active(fs_info->endio_workers, new_pool_size);
workqueue_set_max_active(fs_info->endio_meta_workers, new_pool_size);
btrfs_workqueue_set_max(fs_info->endio_write_workers, new_pool_size);
- btrfs_workqueue_set_max(fs_info->endio_freespace_worker, new_pool_size);
btrfs_workqueue_set_max(fs_info->delayed_workers, new_pool_size);
}
@@ -1264,8 +1240,6 @@ static inline void btrfs_remount_begin(struct btrfs_fs_info *fs_info,
static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info,
unsigned long long old_opts)
{
- const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE);
-
/*
* We need to cleanup all defraggable inodes if the autodefragment is
* close or the filesystem is read only.
@@ -1282,10 +1256,6 @@ static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info,
else if (btrfs_raw_test_opt(old_opts, DISCARD_ASYNC) &&
!btrfs_test_opt(fs_info, DISCARD_ASYNC))
btrfs_discard_cleanup(fs_info);
-
- /* If we toggled space cache */
- if (cache_opt != btrfs_free_space_cache_v1_active(fs_info))
- btrfs_set_free_space_cache_v1_active(fs_info, cache_opt);
}
static int btrfs_remount_rw(struct btrfs_fs_info *fs_info)
@@ -1448,7 +1418,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info,
btrfs_info_if_set(info, old, DISCARD_SYNC, "turning on sync discard");
btrfs_info_if_set(info, old, DISCARD_ASYNC, "turning on async discard");
btrfs_info_if_set(info, old, FREE_SPACE_TREE, "enabling free space tree");
- btrfs_info_if_set(info, old, SPACE_CACHE, "enabling disk space caching");
btrfs_info_if_set(info, old, CLEAR_CACHE, "force clearing of disk cache");
btrfs_info_if_set(info, old, AUTO_DEFRAG, "enabling auto defrag");
btrfs_info_if_set(info, old, FRAGMENT_DATA, "fragmenting data");
@@ -1466,7 +1435,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info,
btrfs_info_if_unset(info, old, SSD_SPREAD, "not using spread ssd allocation scheme");
btrfs_info_if_unset(info, old, NOBARRIER, "turning on barriers");
btrfs_info_if_unset(info, old, NOTREELOG, "enabling tree log");
- btrfs_info_if_unset(info, old, SPACE_CACHE, "disabling disk space caching");
btrfs_info_if_unset(info, old, FREE_SPACE_TREE, "disabling free space tree");
btrfs_info_if_unset(info, old, AUTO_DEFRAG, "disabling auto defrag");
btrfs_info_if_unset(info, old, COMPRESS, "use no compression");
@@ -1531,14 +1499,8 @@ static int btrfs_reconfigure(struct fs_context *fc)
btrfs_warn(fs_info,
"remount supports changing free space tree only from RO to RW");
/* Make sure free space cache options match the state on disk. */
- if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) {
+ if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE))
btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE);
- btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE);
- }
- if (btrfs_free_space_cache_v1_active(fs_info)) {
- btrfs_clear_opt(fs_info->mount_opt, FREE_SPACE_TREE);
- btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE);
- }
}
ret = 0;
diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c
index 39cb01ee441a..c5bb1c7eac6a 100644
--- a/fs/btrfs/sysfs.c
+++ b/fs/btrfs/sysfs.c
@@ -83,8 +83,7 @@ struct raid_kobject {
#define BTRFS_FEAT_ATTR(_name, _feature_set, _feature_prefix, _feature_bit) \
static struct btrfs_feature_attr btrfs_attr_features_##_name = { \
.kobj_attr = __INIT_KOBJ_ATTR(_name, S_IRUGO, \
- btrfs_feature_attr_show, \
- btrfs_feature_attr_store), \
+ btrfs_feature_attr_show, NULL), \
.feature_set = _feature_set, \
.feature_bit = _feature_prefix ##_## _feature_bit, \
}
@@ -130,130 +129,20 @@ static u64 get_features(struct btrfs_fs_info *fs_info,
return btrfs_super_incompat_flags(disk_super);
}
-static void set_features(struct btrfs_fs_info *fs_info,
- enum btrfs_feature_set set, u64 features)
-{
- struct btrfs_super_block *disk_super = fs_info->super_copy;
- if (set == FEAT_COMPAT)
- btrfs_set_super_compat_flags(disk_super, features);
- else if (set == FEAT_COMPAT_RO)
- btrfs_set_super_compat_ro_flags(disk_super, features);
- else
- btrfs_set_super_incompat_flags(disk_super, features);
-}
-
-static int can_modify_feature(struct btrfs_feature_attr *fa)
-{
- int val = 0;
- u64 set, clear;
- switch (fa->feature_set) {
- case FEAT_COMPAT:
- set = BTRFS_FEATURE_COMPAT_SAFE_SET;
- clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR;
- break;
- case FEAT_COMPAT_RO:
- set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET;
- clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR;
- break;
- case FEAT_INCOMPAT:
- set = BTRFS_FEATURE_INCOMPAT_SAFE_SET;
- clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR;
- break;
- default:
- btrfs_warn(NULL, "sysfs: unknown feature set %d", fa->feature_set);
- return 0;
- }
-
- if (set & fa->feature_bit)
- val |= 1;
- if (clear & fa->feature_bit)
- val |= 2;
-
- return val;
-}
-
static ssize_t btrfs_feature_attr_show(struct kobject *kobj,
struct kobj_attribute *a, char *buf)
{
int val = 0;
struct btrfs_fs_info *fs_info = to_fs_info(kobj);
struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a);
+
if (fs_info) {
u64 features = get_features(fs_info, fa->feature_set);
if (features & fa->feature_bit)
val = 1;
- } else
- val = can_modify_feature(fa);
-
- return sysfs_emit(buf, "%d\n", val);
-}
-
-static ssize_t btrfs_feature_attr_store(struct kobject *kobj,
- struct kobj_attribute *a,
- const char *buf, size_t count)
-{
- struct btrfs_fs_info *fs_info;
- struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a);
- u64 features, set, clear;
- unsigned long val;
- int ret;
-
- fs_info = to_fs_info(kobj);
- if (!fs_info)
- return -EPERM;
-
- if (sb_rdonly(fs_info->sb))
- return -EROFS;
-
- ret = kstrtoul(skip_spaces(buf), 0, &val);
- if (ret)
- return ret;
-
- if (fa->feature_set == FEAT_COMPAT) {
- set = BTRFS_FEATURE_COMPAT_SAFE_SET;
- clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR;
- } else if (fa->feature_set == FEAT_COMPAT_RO) {
- set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET;
- clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR;
- } else {
- set = BTRFS_FEATURE_INCOMPAT_SAFE_SET;
- clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR;
- }
-
- features = get_features(fs_info, fa->feature_set);
-
- /* Nothing to do */
- if ((val && (features & fa->feature_bit)) ||
- (!val && !(features & fa->feature_bit)))
- return count;
-
- if ((val && !(set & fa->feature_bit)) ||
- (!val && !(clear & fa->feature_bit))) {
- btrfs_info(fs_info,
- "%sabling feature %s on mounted fs is not supported.",
- val ? "En" : "Dis", fa->kobj_attr.attr.name);
- return -EPERM;
}
- btrfs_info(fs_info, "%s %s feature flag",
- val ? "Setting" : "Clearing", fa->kobj_attr.attr.name);
-
- spin_lock(&fs_info->super_lock);
- features = get_features(fs_info, fa->feature_set);
- if (val)
- features |= fa->feature_bit;
- else
- features &= ~fa->feature_bit;
- set_features(fs_info, fa->feature_set, features);
- spin_unlock(&fs_info->super_lock);
-
- /*
- * We don't want to do full transaction commit from inside sysfs
- */
- set_bit(BTRFS_FS_NEED_TRANS_COMMIT, &fs_info->flags);
- wake_up_process(fs_info->transaction_kthread);
-
- return count;
+ return sysfs_emit(buf, "%d\n", val);
}
static umode_t btrfs_feature_visible(struct kobject *kobj,
@@ -269,9 +158,7 @@ static umode_t btrfs_feature_visible(struct kobject *kobj,
fa = attr_to_btrfs_feature_attr(attr);
features = get_features(fs_info, fa->feature_set);
- if (can_modify_feature(fa))
- mode |= S_IWUSR;
- else if (!(features & fa->feature_bit))
+ if (!(features & fa->feature_bit))
mode = 0;
}
@@ -2359,9 +2246,7 @@ static ssize_t qgroup_enabled_show(struct kobject *qgroups_kobj,
struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent);
bool enabled;
- spin_lock(&fs_info->qgroup_lock);
- enabled = fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON;
- spin_unlock(&fs_info->qgroup_lock);
+ enabled = test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags);
return sysfs_emit(buf, "%d\n", enabled);
}
@@ -2401,9 +2286,7 @@ static ssize_t qgroup_inconsistent_show(struct kobject *qgroups_kobj,
struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent);
bool inconsistent;
- spin_lock(&fs_info->qgroup_lock);
- inconsistent = (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT);
- spin_unlock(&fs_info->qgroup_lock);
+ inconsistent = test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags);
return sysfs_emit(buf, "%d\n", inconsistent);
}
diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c
index 23459cd4e503..cd045778400d 100644
--- a/fs/btrfs/tests/extent-io-tests.c
+++ b/fs/btrfs/tests/extent-io-tests.c
@@ -18,8 +18,8 @@
#define PROCESS_RELEASE (1U << 1)
#define PROCESS_TEST_LOCKED (1U << 2)
-static noinline int process_page_range(struct inode *inode, u64 start, u64 end,
- unsigned long flags)
+static noinline int process_folio_range(struct inode *inode, u64 start, u64 end,
+ unsigned long flags)
{
int ret;
struct folio_batch fbatch;
@@ -112,8 +112,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
struct btrfs_root *root = NULL;
struct inode *inode = NULL;
struct extent_io_tree *tmp;
- struct page *page;
- struct page *locked_page = NULL;
+ struct folio *folio;
+ struct folio *locked_folio = NULL;
/* In this test we need at least 2 file extents at its maximum size */
u64 max_bytes = BTRFS_MAX_EXTENT_SIZE;
u64 total_dirty = 2 * max_bytes;
@@ -152,23 +152,27 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
btrfs_extent_io_tree_init(NULL, tmp, IO_TREE_SELFTEST);
/*
- * First go through and create and mark all of our pages dirty, we pin
- * everything to make sure our pages don't get evicted and screw up our
+ * First go through and create and mark all of our folios dirty, we pin
+ * everything to make sure our folios don't get evicted and screw up our
* test.
*/
for (pgoff_t index = 0; index < (total_dirty >> PAGE_SHIFT); index++) {
- page = find_or_create_page(inode->i_mapping, index, GFP_KERNEL);
- if (!page) {
- test_err("failed to allocate test page");
- ret = -ENOMEM;
+ folio = __filemap_get_folio(inode->i_mapping, index,
+ FGP_LOCK | FGP_ACCESSED | FGP_CREAT,
+ GFP_KERNEL);
+ if (IS_ERR(folio)) {
+ test_err("failed to allocate test folio");
+ ret = PTR_ERR(folio);
goto out;
}
- SetPageDirty(page);
+ /* The ranges below assume page sized folios. */
+ ASSERT(folio_order(folio) == 0);
+ folio_set_dirty(folio);
if (index) {
- unlock_page(page);
+ folio_unlock(folio);
} else {
- get_page(page);
- locked_page = page;
+ folio_get(folio);
+ locked_folio = folio;
}
}
@@ -179,8 +183,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
btrfs_set_extent_bit(tmp, 0, sectorsize - 1, EXTENT_DELALLOC, NULL);
start = 0;
end = start + PAGE_SIZE - 1;
- found = find_lock_delalloc_range(inode, page_folio(locked_page), &start,
- &end);
+ found = find_lock_delalloc_range(inode, locked_folio, &start, &end);
if (!found) {
test_err("should have found at least one delalloc");
goto out_bits;
@@ -191,8 +194,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
goto out_bits;
}
btrfs_unlock_extent(tmp, start, end, NULL);
- unlock_page(locked_page);
- put_page(locked_page);
+ folio_unlock(locked_folio);
+ folio_put(locked_folio);
/*
* Test this scenario
@@ -201,17 +204,17 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
* |--- search ---|
*/
test_start = SZ_64M;
- locked_page = find_lock_page(inode->i_mapping,
- test_start >> PAGE_SHIFT);
- if (!locked_page) {
- test_err("couldn't find the locked page");
+ locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT);
+ if (IS_ERR(locked_folio)) {
+ test_err("couldn't find the locked folio");
+ locked_folio = NULL;
goto out_bits;
}
+ ASSERT(folio_order(locked_folio) == 0);
btrfs_set_extent_bit(tmp, sectorsize, max_bytes - 1, EXTENT_DELALLOC, NULL);
start = test_start;
end = start + PAGE_SIZE - 1;
- found = find_lock_delalloc_range(inode, page_folio(locked_page), &start,
- &end);
+ found = find_lock_delalloc_range(inode, locked_folio, &start, &end);
if (!found) {
test_err("couldn't find delalloc in our range");
goto out_bits;
@@ -221,14 +224,14 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
test_start, max_bytes - 1, start, end);
goto out_bits;
}
- if (process_page_range(inode, start, end,
- PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) {
- test_err("there were unlocked pages in the range");
+ if (process_folio_range(inode, start, end,
+ PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) {
+ test_err("there were unlocked folios in the range");
goto out_bits;
}
btrfs_unlock_extent(tmp, start, end, NULL);
- /* locked_page was unlocked above */
- put_page(locked_page);
+ /* locked_folio was unlocked above */
+ folio_put(locked_folio);
/*
* Test this scenario
@@ -236,16 +239,16 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
* |--- search ---|
*/
test_start = max_bytes + sectorsize;
- locked_page = find_lock_page(inode->i_mapping, test_start >>
- PAGE_SHIFT);
- if (!locked_page) {
- test_err("couldn't find the locked page");
+ locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT);
+ if (IS_ERR(locked_folio)) {
+ test_err("couldn't find the locked folio");
+ locked_folio = NULL;
goto out_bits;
}
+ ASSERT(folio_order(locked_folio) == 0);
start = test_start;
end = start + PAGE_SIZE - 1;
- found = find_lock_delalloc_range(inode, page_folio(locked_page), &start,
- &end);
+ found = find_lock_delalloc_range(inode, locked_folio, &start, &end);
if (found) {
test_err("found range when we shouldn't have");
goto out_bits;
@@ -265,8 +268,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
btrfs_set_extent_bit(tmp, max_bytes, total_dirty - 1, EXTENT_DELALLOC, NULL);
start = test_start;
end = start + PAGE_SIZE - 1;
- found = find_lock_delalloc_range(inode, page_folio(locked_page), &start,
- &end);
+ found = find_lock_delalloc_range(inode, locked_folio, &start, &end);
if (!found) {
test_err("didn't find our range");
goto out_bits;
@@ -276,38 +278,37 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
test_start, total_dirty - 1, start, end);
goto out_bits;
}
- if (process_page_range(inode, start, end,
- PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) {
- test_err("pages in range were not all locked");
+ if (process_folio_range(inode, start, end,
+ PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) {
+ test_err("folios in range were not all locked");
goto out_bits;
}
btrfs_unlock_extent(tmp, start, end, NULL);
/*
- * Now to test where we run into a page that is no longer dirty in the
+ * Now to test where we run into a folio that is no longer dirty in the
* range we want to find.
*/
- page = find_get_page(inode->i_mapping,
- (max_bytes + SZ_1M) >> PAGE_SHIFT);
- if (!page) {
- test_err("couldn't find our page");
+ folio = filemap_get_folio(inode->i_mapping, (max_bytes + SZ_1M) >> PAGE_SHIFT);
+ if (IS_ERR(folio)) {
+ test_err("couldn't find our folio");
goto out_bits;
}
- ClearPageDirty(page);
- put_page(page);
+ ASSERT(folio_order(folio) == 0);
+ folio_clear_dirty(folio);
+ folio_put(folio);
/* We unlocked it in the previous test */
- lock_page(locked_page);
+ folio_lock(locked_folio);
start = test_start;
end = start + PAGE_SIZE - 1;
/*
- * Currently if we fail to find dirty pages in the delalloc range we
+ * Currently if we fail to find dirty folios in the delalloc range we
* will adjust max_bytes down to PAGE_SIZE and then re-search. If
* this changes at any point in the future we will need to fix this
* tests expected behavior.
*/
- found = find_lock_delalloc_range(inode, page_folio(locked_page), &start,
- &end);
+ found = find_lock_delalloc_range(inode, locked_folio, &start, &end);
if (!found) {
test_err("didn't find our range");
goto out_bits;
@@ -317,9 +318,9 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
test_start, test_start + PAGE_SIZE - 1, start, end);
goto out_bits;
}
- if (process_page_range(inode, start, end, PROCESS_TEST_LOCKED |
- PROCESS_UNLOCK)) {
- test_err("pages in range were not all locked");
+ if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED |
+ PROCESS_UNLOCK)) {
+ test_err("folios in range were not all locked");
goto out_bits;
}
ret = 0;
@@ -328,10 +329,10 @@ out_bits:
dump_extent_io_tree(tmp);
btrfs_clear_extent_bit(tmp, 0, total_dirty - 1, (unsigned)-1, NULL);
out:
- if (locked_page)
- put_page(locked_page);
- process_page_range(inode, 0, total_dirty - 1,
- PROCESS_UNLOCK | PROCESS_RELEASE);
+ if (locked_folio)
+ folio_put(locked_folio);
+ process_folio_range(inode, 0, total_dirty - 1,
+ PROCESS_UNLOCK | PROCESS_RELEASE);
iput(inode);
out_root_info:
btrfs_free_dummy_root(root);
@@ -671,8 +672,9 @@ static void dump_eb_and_memory_contents(struct extent_buffer *eb, void *memory,
const char *test_name)
{
for (int i = 0; i < eb->len; i++) {
- struct page *page = folio_page(eb->folios[i >> PAGE_SHIFT], 0);
- void *addr = page_address(page) + offset_in_page(i);
+ const unsigned long idx = get_eb_folio_index(eb, i);
+ void *addr = folio_address(eb->folios[idx]) +
+ get_eb_offset_in_folio(eb, i);
if (memcmp(addr, memory + i, 1) != 0) {
test_err("%s failed", test_name);
@@ -687,9 +689,12 @@ static int verify_eb_and_memory(struct extent_buffer *eb, void *memory,
const char *test_name)
{
for (int i = 0; i < (eb->len >> PAGE_SHIFT); i++) {
- void *eb_addr = folio_address(eb->folios[i]);
+ const unsigned long offset = i << PAGE_SHIFT;
+ const unsigned long idx = get_eb_folio_index(eb, offset);
+ void *eb_addr = folio_address(eb->folios[idx]) +
+ get_eb_offset_in_folio(eb, offset);
- if (memcmp(memory + (i << PAGE_SHIFT), eb_addr, PAGE_SIZE) != 0) {
+ if (memcmp(memory + offset, eb_addr, PAGE_SIZE) != 0) {
dump_eb_and_memory_contents(eb, memory, test_name);
return -EUCLEAN;
}
diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c
index 6802b94ed76f..84f012bfcffc 100644
--- a/fs/btrfs/transaction.c
+++ b/fs/btrfs/transaction.c
@@ -125,17 +125,14 @@ static const unsigned int btrfs_blocked_trans_types[TRANS_STATE_MAX] = {
[TRANS_STATE_UNBLOCKED] = (__TRANS_START |
__TRANS_ATTACH |
__TRANS_JOIN |
- __TRANS_JOIN_NOLOCK |
__TRANS_JOIN_NOSTART),
[TRANS_STATE_SUPER_COMMITTED] = (__TRANS_START |
__TRANS_ATTACH |
__TRANS_JOIN |
- __TRANS_JOIN_NOLOCK |
__TRANS_JOIN_NOSTART),
[TRANS_STATE_COMPLETED] = (__TRANS_START |
__TRANS_ATTACH |
__TRANS_JOIN |
- __TRANS_JOIN_NOLOCK |
__TRANS_JOIN_NOSTART),
};
@@ -310,12 +307,6 @@ loop:
if (type == TRANS_ATTACH || type == TRANS_JOIN_NOSTART)
return -ENOENT;
- /*
- * JOIN_NOLOCK only happens during the transaction commit, so
- * it is impossible that ->running_transaction is NULL
- */
- BUG_ON(type == TRANS_JOIN_NOLOCK);
-
cur_trans = kmalloc_obj(*cur_trans, GFP_NOFS);
if (!cur_trans)
return -ENOMEM;
@@ -379,9 +370,8 @@ loop:
INIT_LIST_HEAD(&cur_trans->dev_update_list);
INIT_LIST_HEAD(&cur_trans->switch_commits);
INIT_LIST_HEAD(&cur_trans->dirty_bgs);
- INIT_LIST_HEAD(&cur_trans->io_bgs);
INIT_LIST_HEAD(&cur_trans->dropped_roots);
- mutex_init(&cur_trans->cache_write_mutex);
+ mutex_init(&cur_trans->dirty_bgs_update_mutex);
spin_lock_init(&cur_trans->dirty_bgs_lock);
INIT_LIST_HEAD(&cur_trans->deleted_bgs);
spin_lock_init(&cur_trans->dropped_roots_lock);
@@ -710,14 +700,8 @@ again:
}
/*
- * If we are JOIN_NOLOCK we're already committing a transaction and
- * waiting on this guy, so we don't need to do the sb_start_intwrite
- * because we're already holding a ref. We need this because we could
- * have raced in and did an fsync() on a file which can kick a commit
- * and then we deadlock with somebody doing a freeze.
- *
* If we are ATTACH, it means we just want to catch the current
- * transaction and commit it, so we needn't do sb_start_intwrite().
+ * transaction and commit it, so we needn't do sb_start_intwrite().
*/
if (type & __TRANS_FREEZABLE)
sb_start_intwrite(fs_info->sb);
@@ -855,12 +839,6 @@ struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root)
true);
}
-struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root)
-{
- return start_transaction(root, 0, TRANS_JOIN_NOLOCK,
- BTRFS_RESERVE_NO_FLUSH, true);
-}
-
/*
* Similar to regular join but it never starts a transaction when none is
* running or when there's a running one at a state >= TRANS_STATE_UNBLOCKED.
@@ -1363,7 +1341,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans)
{
struct btrfs_fs_info *fs_info = trans->fs_info;
struct list_head *dirty_bgs = &trans->transaction->dirty_bgs;
- struct list_head *io_bgs = &trans->transaction->io_bgs;
struct extent_buffer *eb;
int ret;
@@ -1393,10 +1370,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans)
if (ret)
return ret;
- ret = btrfs_setup_space_cache(trans);
- if (ret)
- return ret;
-
again:
while (!list_empty(&fs_info->dirty_cowonly_roots)) {
struct btrfs_root *root;
@@ -1417,7 +1390,7 @@ again:
if (ret)
return ret;
- while (!list_empty(dirty_bgs) || !list_empty(io_bgs)) {
+ while (!list_empty(dirty_bgs)) {
ret = btrfs_write_dirty_block_groups(trans);
if (ret)
return ret;
@@ -1890,8 +1863,7 @@ static noinline int create_pending_snapshot(struct btrfs_trans_handle *trans,
goto fail;
ret = btrfs_insert_dir_item(trans, &fname.disk_name,
- parent_inode, &key, BTRFS_FT_DIR,
- index);
+ parent_inode, &key, BTRFS_FT_DIR, index, NULL);
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
goto fail;
@@ -1990,9 +1962,7 @@ static void update_super_roots(struct btrfs_fs_info *fs_info)
super->root = root_item->bytenr;
super->generation = root_item->generation;
super->root_level = root_item->level;
- if (btrfs_test_opt(fs_info, SPACE_CACHE))
- super->cache_generation = root_item->generation;
- else if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags))
+ if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags))
super->cache_generation = 0;
if (test_bit(BTRFS_FS_UPDATE_UUID_TREE_GEN, &fs_info->flags))
super->uuid_tree_generation = root_item->generation;
@@ -2274,18 +2244,16 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans)
if (!test_bit(BTRFS_TRANS_DIRTY_BG_RUN, &cur_trans->flags)) {
bool run_it = false;
- /* this mutex is also taken before trying to set
- * block groups readonly. We need to make sure
- * that nobody has set a block group readonly
- * after a extents from that block group have been
- * allocated for cache files. btrfs_set_block_group_ro
- * will wait for the transaction to commit if it
- * finds BTRFS_TRANS_DIRTY_BG_RUN set.
+ /*
+ * This mutex is also taken before trying to set block groups
+ * readonly. btrfs_inc_block_group_ro() will wait for the
+ * transaction to commit if it finds BTRFS_TRANS_DIRTY_BG_RUN
+ * set.
*
* The BTRFS_TRANS_DIRTY_BG_RUN flag is also used to make sure
- * only one process starts all the block group IO. It wouldn't
- * hurt to have more than one go through, but there's no
- * real advantage to it either.
+ * only one process starts all the block group item updates. It
+ * wouldn't hurt to have more than one go through, but there's
+ * no real advantage to it either.
*/
mutex_lock(&fs_info->ro_block_group_mutex);
if (!test_and_set_bit(BTRFS_TRANS_DIRTY_BG_RUN,
@@ -2519,10 +2487,7 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans)
if (unlikely(ret))
goto unlock_reloc;
- /*
- * The tasks which save the space cache and inode cache may also
- * update ->aborted, check it.
- */
+ /* Other tasks may also have updated ->aborted, check it. */
if (TRANS_ABORTED(cur_trans)) {
ret = cur_trans->aborted;
goto unlock_reloc;
@@ -2543,7 +2508,6 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans)
switch_commit_roots(trans);
ASSERT(list_empty(&cur_trans->dirty_bgs));
- ASSERT(list_empty(&cur_trans->io_bgs));
update_super_roots(fs_info);
btrfs_set_super_log_root(fs_info->super_copy, 0);
diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h
index 3a57f227b5ed..68c724e70809 100644
--- a/fs/btrfs/transaction.h
+++ b/fs/btrfs/transaction.h
@@ -47,7 +47,6 @@ enum btrfs_trans_state {
#define BTRFS_TRANS_HAVE_FREE_BGS 0
#define BTRFS_TRANS_DIRTY_BG_RUN 1
-#define BTRFS_TRANS_CACHE_ENOSPC 2
struct btrfs_transaction {
u64 transid;
@@ -78,32 +77,15 @@ struct btrfs_transaction {
struct list_head dev_update_list;
struct list_head switch_commits;
struct list_head dirty_bgs;
-
- /*
- * There is no explicit lock which protects io_bgs, rather its
- * consistency is implied by the fact that all the sites which modify
- * it do so under some form of transaction critical section, namely:
- *
- * - btrfs_start_dirty_block_groups - This function can only ever be
- * run by one of the transaction committers. Refer to
- * BTRFS_TRANS_DIRTY_BG_RUN usage in btrfs_commit_transaction
- *
- * - btrfs_write_dirty_blockgroups - this is called by
- * commit_cowonly_roots from transaction critical section
- * (TRANS_STATE_COMMIT_DOING)
- *
- * - btrfs_cleanup_dirty_bgs - called on transaction abort
- */
- struct list_head io_bgs;
struct list_head dropped_roots;
struct extent_io_tree pinned_extents;
/*
- * we need to make sure block group deletion doesn't race with
- * free space cache writeout. This mutex keeps them from stomping
- * on each other
+ * We need to make sure block group deletion doesn't race with the
+ * dirty block group item updates done outside the commit critical
+ * section. This mutex keeps them from stomping on each other.
*/
- struct mutex cache_write_mutex;
+ struct mutex dirty_bgs_update_mutex;
spinlock_t dirty_bgs_lock;
/* Protected by spin lock fs_info->unused_bgs_lock. */
struct list_head deleted_bgs;
@@ -124,7 +106,6 @@ enum {
ENUM_BIT(__TRANS_START),
ENUM_BIT(__TRANS_ATTACH),
ENUM_BIT(__TRANS_JOIN),
- ENUM_BIT(__TRANS_JOIN_NOLOCK),
ENUM_BIT(__TRANS_DUMMY),
ENUM_BIT(__TRANS_JOIN_NOSTART),
};
@@ -132,7 +113,6 @@ enum {
#define TRANS_START (__TRANS_START | __TRANS_FREEZABLE)
#define TRANS_ATTACH (__TRANS_ATTACH)
#define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE)
-#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK)
#define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART)
#define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH)
@@ -288,7 +268,7 @@ do { \
* Call btrfs_abort_transaction() as early as possible when an error condition
* is detected, that way the exact stack trace is reported for some errors.
*
- * Error number must be negative as it encodes wheather it's the first abort.
+ * Error number must be negative as it encodes whether it's the first abort.
*/
#define btrfs_abort_transaction(trans, error) \
do { \
@@ -312,7 +292,6 @@ struct btrfs_trans_handle *btrfs_start_transaction_fallback_global_rsv(
struct btrfs_root *root,
unsigned int num_items);
struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root);
-struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root);
struct btrfs_trans_handle *btrfs_join_transaction_nostart(struct btrfs_root *root);
struct btrfs_trans_handle *btrfs_attach_transaction(struct btrfs_root *root);
struct btrfs_trans_handle *btrfs_attach_transaction_barrier(
diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c
index 4b1e47173c63..9cd79d97b9b5 100644
--- a/fs/btrfs/tree-checker.c
+++ b/fs/btrfs/tree-checker.c
@@ -108,13 +108,14 @@ static void file_extent_err(const struct extent_buffer *eb, int slot,
*/
#define CHECK_FE_ALIGNED(leaf, slot, fi, name, alignment) \
({ \
- if (unlikely(!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), \
- (alignment)))) \
+ const u64 val = btrfs_file_extent_##name((leaf), (fi)); \
+ const bool not_aligned = !IS_ALIGNED(val, (alignment)); \
+ \
+ if (unlikely(not_aligned)) \
file_extent_err((leaf), (slot), \
"invalid %s for file extent, have %llu, should be aligned to %u", \
- (#name), btrfs_file_extent_##name((leaf), (fi)), \
- (alignment)); \
- (!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), (alignment))); \
+ (#name), val, (alignment)); \
+ not_aligned; \
})
static u64 file_extent_end(struct extent_buffer *leaf,
@@ -163,6 +164,12 @@ static void dir_item_err(const struct extent_buffer *eb, int slot,
va_end(args);
}
+/* Record info for the last hit inode. */
+struct saved_inode_info {
+ u64 ino;
+ u32 mode;
+};
+
/*
* This functions checks prev_key->objectid, to ensure current key and prev_key
* share the same objectid as inode number.
@@ -204,15 +211,41 @@ static bool check_prev_ino(struct extent_buffer *leaf,
prev_key->objectid, key->objectid);
return false;
}
+
+static bool can_have_extent_data(struct extent_buffer *leaf,
+ struct btrfs_key *key, int slot, u8 fi_type,
+ const struct saved_inode_info *inode_info)
+{
+ /* No inode item in this leaf. */
+ if (inode_info->ino != key->objectid)
+ return true;
+ if (S_ISREG(inode_info->mode))
+ return true;
+ if (S_ISLNK(inode_info->mode)) {
+ /* For a symlink, the file extent item should always be inlined. */
+ if (unlikely(fi_type != BTRFS_FILE_EXTENT_INLINE))
+ return false;
+ return true;
+ }
+
+ /*
+ * The rest are special files, e.g. block/FIFO files, which cannot
+ * have any file extent.
+ */
+ return false;
+}
+
static int check_extent_data_item(struct extent_buffer *leaf,
struct btrfs_key *key, int slot,
- struct btrfs_key *prev_key)
+ struct btrfs_key *prev_key,
+ const struct saved_inode_info *inode_info)
{
struct btrfs_fs_info *fs_info = leaf->fs_info;
struct btrfs_file_extent_item *fi;
u32 sectorsize = fs_info->sectorsize;
u32 item_size = btrfs_item_size(leaf, slot);
u64 extent_end;
+ u8 fi_type;
if (unlikely(!IS_ALIGNED(key->offset, sectorsize))) {
file_extent_err(leaf, slot,
@@ -243,12 +276,18 @@ static int check_extent_data_item(struct extent_buffer *leaf,
SZ_4K);
return -EUCLEAN;
}
- if (unlikely(btrfs_file_extent_type(leaf, fi) >=
- BTRFS_NR_FILE_EXTENT_TYPES)) {
+ fi_type = btrfs_file_extent_type(leaf, fi);
+ if (unlikely(fi_type >= BTRFS_NR_FILE_EXTENT_TYPES)) {
file_extent_err(leaf, slot,
"invalid type for file extent, have %u expect range [0, %u]",
- btrfs_file_extent_type(leaf, fi),
- BTRFS_NR_FILE_EXTENT_TYPES - 1);
+ fi_type, BTRFS_NR_FILE_EXTENT_TYPES - 1);
+ return -EUCLEAN;
+ }
+
+ if (unlikely(!can_have_extent_data(leaf, key, slot, fi_type, inode_info))) {
+ file_extent_err(leaf, slot,
+ "unexpected file extent item type %u for inode mode 0%o",
+ fi_type, inode_info->mode);
return -EUCLEAN;
}
@@ -270,7 +309,8 @@ static int check_extent_data_item(struct extent_buffer *leaf,
btrfs_file_extent_encryption(leaf, fi));
return -EUCLEAN;
}
- if (btrfs_file_extent_type(leaf, fi) == BTRFS_FILE_EXTENT_INLINE) {
+
+ if (fi_type == BTRFS_FILE_EXTENT_INLINE) {
/* Inline extent must have 0 as key offset */
if (unlikely(key->offset)) {
file_extent_err(leaf, slot,
@@ -1081,9 +1121,9 @@ int btrfs_check_chunk_valid(const struct btrfs_fs_info *fs_info,
return -EUCLEAN;
}
- if (!remapped &&
- !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK,
- num_stripes, sub_stripes)) {
+ if (unlikely(!remapped &&
+ !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK,
+ num_stripes, sub_stripes))) {
chunk_err(fs_info, leaf, chunk, logical,
"invalid num_stripes:sub_stripes %u:%u for profile %llu",
num_stripes, sub_stripes,
@@ -1206,7 +1246,8 @@ static int check_dev_item(struct extent_buffer *leaf,
}
static int check_inode_item(struct extent_buffer *leaf,
- struct btrfs_key *key, int slot)
+ struct btrfs_key *key, int slot,
+ struct saved_inode_info *inode_info)
{
struct btrfs_fs_info *fs_info = leaf->fs_info;
struct btrfs_inode_item *iitem;
@@ -1291,6 +1332,8 @@ static int check_inode_item(struct extent_buffer *leaf,
ro_flags);
return -EUCLEAN;
}
+ inode_info->ino = key->objectid;
+ inode_info->mode = mode;
return 0;
}
@@ -2348,14 +2391,15 @@ static int check_free_space_bitmap(struct extent_buffer *leaf,
static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf,
struct btrfs_key *key,
int slot,
- struct btrfs_key *prev_key)
+ struct btrfs_key *prev_key,
+ struct saved_inode_info *inode_info)
{
int ret = 0;
struct btrfs_chunk *chunk;
switch (key->type) {
case BTRFS_EXTENT_DATA_KEY:
- ret = check_extent_data_item(leaf, key, slot, prev_key);
+ ret = check_extent_data_item(leaf, key, slot, prev_key, inode_info);
break;
case BTRFS_EXTENT_CSUM_KEY:
ret = check_csum_item(leaf, key, slot, prev_key);
@@ -2385,7 +2429,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf,
ret = check_dev_extent_item(leaf, key, slot, prev_key);
break;
case BTRFS_INODE_ITEM_KEY:
- ret = check_inode_item(leaf, key, slot);
+ ret = check_inode_item(leaf, key, slot, inode_info);
break;
case BTRFS_ROOT_ITEM_KEY:
ret = check_root_item(leaf, key, slot);
@@ -2433,6 +2477,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf,
enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf)
{
struct btrfs_fs_info *fs_info = leaf->fs_info;
+ struct saved_inode_info inode_info = { 0 };
/* No valid key type is 0, so all key should be larger than this key */
struct btrfs_key prev_key = {0, 0, 0};
struct btrfs_key key;
@@ -2568,7 +2613,7 @@ enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf)
}
/* Check if the item size and content meet other criteria. */
- ret = check_leaf_item(leaf, &key, slot, &prev_key);
+ ret = check_leaf_item(leaf, &key, slot, &prev_key, &inode_info);
if (unlikely(ret != BTRFS_TREE_BLOCK_CLEAN))
return ret;
diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c
index a00094604e54..1d9dfb63fc5a 100644
--- a/fs/btrfs/tree-log.c
+++ b/fs/btrfs/tree-log.c
@@ -503,7 +503,7 @@ static int overwrite_item(struct walk_control *wc)
btrfs_release_path(wc->subvol_path);
return 0;
}
- src_copy = kmalloc(item_size, GFP_NOFS);
+ src_copy = kvmalloc(item_size, GFP_NOFS);
if (!src_copy) {
btrfs_abort_log_replay(wc, -ENOMEM,
"failed to allocate memory for log leaf item");
@@ -514,7 +514,7 @@ static int overwrite_item(struct walk_control *wc)
dst_ptr = btrfs_item_ptr_offset(dst_eb, dst_slot);
ret = memcmp_extent_buffer(dst_eb, src_copy, dst_ptr, item_size);
- kfree(src_copy);
+ kvfree(src_copy);
/*
* they have the same contents, just return, this saves
* us from cowing blocks in the destination tree and doing
@@ -1683,7 +1683,7 @@ static noinline int add_inode_ref(struct walk_control *wc)
}
/* insert our name */
- ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index);
+ ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index, NULL);
if (ret) {
btrfs_abort_log_replay(wc, ret,
"failed to add link for inode %llu in dir %llu ref_index %llu name %.*s root %llu",
@@ -2031,7 +2031,7 @@ static noinline int insert_one_name(struct btrfs_trans_handle *trans,
return PTR_ERR(dir);
}
- ret = btrfs_add_link(trans, dir, inode, name, true, index);
+ ret = btrfs_add_link(trans, dir, inode, name, true, index, NULL);
/* FIXME, put inode into FIXUP list */
@@ -5350,7 +5350,7 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans,
* have a bunch of extents we just want to commit since it will
* be faster.
*/
- if (++num > 32768) {
+ if (++num > SZ_16K) {
list_del_init(&tree->modified_extents);
ret = -EFBIG;
goto process;
@@ -5368,7 +5368,6 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans,
refcount_inc(&em->refs);
em->flags |= EXTENT_FLAG_LOGGING;
list_add_tail(&em->list, &extents);
- num++;
}
list_sort(NULL, &extents, extent_cmp);
diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c
index 600337a84fbe..83dc7dee14cf 100644
--- a/fs/btrfs/verity.c
+++ b/fs/btrfs/verity.c
@@ -286,21 +286,17 @@ static int write_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset,
* @dest: Buffer to read into. This parameter has slightly tricky
* semantics. If it is NULL, the function will not do any copying
* and will just return the size of all the items up to len bytes.
- * If dest_page is passed, then the function will kmap_local the
- * page and ignore dest, but it must still be non-NULL to avoid the
- * counting-only behavior.
* @len: length in bytes to read
- * @dest_folio: copy into this folio instead of the dest buffer
*
* Helper function to read items from the btree. This returns the number of
* bytes read or < 0 for errors. We can return short reads if the items don't
* exist on disk or aren't big enough to fill the desired length. Supports
- * reading into a provided buffer (dest) or into the page cache
+ * reading into a provided buffer (dest).
*
* Returns number of bytes read or a negative error code on failure.
*/
static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset,
- char *dest, u64 len, struct folio *dest_folio)
+ char *dest, u64 len)
{
BTRFS_PATH_AUTO_FREE(path);
struct btrfs_root *root = inode->root;
@@ -320,7 +316,11 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset,
if (!path)
return -ENOMEM;
- if (dest_folio)
+ /*
+ * Merkle items can be large and split across multiple items, so enable
+ * readahead for such cases.
+ */
+ if (key_type == BTRFS_VERITY_MERKLE_ITEM_KEY)
path->reada = READA_FORWARD;
key.objectid = btrfs_ino(inode);
@@ -364,7 +364,7 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset,
break;
}
- /* desc = NULL to just sum all the item lengths */
+ /* dest == NULL to just sum all the item lengths */
if (!dest)
copy_end = item_end;
else
@@ -377,16 +377,10 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset,
copy_offset = offset - key.offset;
if (dest) {
- if (dest_folio)
- kaddr = kmap_local_folio(dest_folio, 0);
-
data = btrfs_item_ptr(leaf, path->slots[0], void);
read_extent_buffer(leaf, kaddr + dest_offset,
(unsigned long)data + copy_offset,
copy_bytes);
-
- if (dest_folio)
- kunmap_local(kaddr);
}
offset += copy_bytes;
@@ -677,7 +671,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size)
memset(&item, 0, sizeof(item));
ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 0,
- (char *)&item, sizeof(item), NULL);
+ (char *)&item, sizeof(item));
if (ret < 0)
return ret;
@@ -694,7 +688,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size)
return -ERANGE;
ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 1,
- buf, buf_size, NULL);
+ buf, buf_size);
if (ret < 0)
return ret;
if (ret != true_size)
@@ -720,6 +714,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode,
struct folio *folio;
u64 off = (u64)index << PAGE_SHIFT;
loff_t merkle_pos = merkle_file_pos(inode);
+ void *kaddr;
int ret;
if (merkle_pos < 0)
@@ -763,6 +758,7 @@ again:
}
read_folio:
+ kaddr = kmap_local_folio(folio, 0);
/*
* Merkle item keys are indexed from byte 0 in the merkle tree.
* They have the form:
@@ -770,7 +766,8 @@ read_folio:
* [ inode objectid, BTRFS_MERKLE_ITEM_KEY, offset in bytes ]
*/
ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_MERKLE_ITEM_KEY, off,
- folio_address(folio), PAGE_SIZE, folio);
+ kaddr, PAGE_SIZE);
+ kunmap_local(kaddr);
if (ret < 0) {
folio_unlock(folio);
folio_put(folio);
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
index 85ea9c5d4536..7c040f22dbc3 100644
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -403,7 +403,7 @@ static struct btrfs_fs_devices *alloc_fs_devices(const u8 *fsid)
return fs_devs;
}
-static void btrfs_free_device(struct btrfs_device *device)
+void btrfs_free_device(struct btrfs_device *device)
{
WARN_ON(!list_empty(&device->post_commit_list));
/*
@@ -1358,16 +1358,14 @@ int btrfs_open_devices(struct btrfs_fs_devices *fs_devices,
void btrfs_release_disk_super(struct btrfs_super_block *super)
{
- struct page *page = virt_to_page(super);
-
- put_page(page);
+ folio_put(virt_to_folio(super));
}
struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev,
int copy_num, bool drop_cache)
{
struct btrfs_super_block *super;
- struct page *page;
+ struct folio *folio;
u64 bytenr, bytenr_orig;
struct address_space *mapping = bdev->bd_mapping;
int ret;
@@ -1388,7 +1386,7 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev,
ASSERT(copy_num == 0);
/*
- * Drop the page of the primary superblock, so later read will
+ * Drop the folio of the primary superblock, so later read will
* always read from the device.
*/
invalidate_inode_pages2_range(mapping, bytenr >> PAGE_SHIFT,
@@ -1396,12 +1394,12 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev,
}
filemap_invalidate_lock_shared(mapping);
- page = read_cache_page_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS);
+ folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS);
filemap_invalidate_unlock_shared(mapping);
- if (IS_ERR(page))
- return ERR_CAST(page);
+ if (IS_ERR(folio))
+ return ERR_CAST(folio);
- super = page_address(page);
+ super = folio_address(folio) + offset_in_folio(folio, bytenr);
if (btrfs_super_magic(super) != BTRFS_MAGIC ||
btrfs_super_bytenr(super) != bytenr_orig) {
btrfs_release_disk_super(super);
@@ -2814,6 +2812,41 @@ static void btrfs_setup_sprout(struct btrfs_fs_info *fs_info,
btrfs_set_super_flags(disk_super, super_flags);
}
+static void btrfs_rollback_sprout(struct btrfs_fs_info *fs_info,
+ struct btrfs_fs_devices *seed_devices)
+{
+ struct btrfs_fs_devices *fs_devices = fs_info->fs_devices;
+ struct btrfs_super_block *disk_super = fs_info->super_copy;
+ struct btrfs_device *device;
+ u64 super_flags;
+
+ lockdep_assert_held(&uuid_mutex);
+ lockdep_assert_held(&fs_devices->device_list_mutex);
+
+ list_del_init(&seed_devices->seed_list);
+ list_splice_init_rcu(&seed_devices->devices, &fs_devices->devices, synchronize_rcu);
+ list_for_each_entry(device, &fs_devices->devices, dev_list) {
+ device->fs_devices = fs_devices;
+ }
+
+ fs_devices->seeding = true;
+ fs_devices->num_devices = seed_devices->num_devices;
+ fs_devices->open_devices = seed_devices->open_devices;
+ fs_devices->missing_devices = seed_devices->missing_devices;
+ fs_devices->rotating = seed_devices->rotating;
+ fs_devices->latest_dev = seed_devices->latest_dev;
+
+ memcpy(fs_devices->fsid, seed_devices->fsid, BTRFS_FSID_SIZE);
+ memcpy(fs_devices->metadata_uuid, seed_devices->metadata_uuid, BTRFS_FSID_SIZE);
+ memcpy(disk_super->fsid, seed_devices->fsid, BTRFS_FSID_SIZE);
+
+ super_flags = (btrfs_super_flags(disk_super) | BTRFS_SUPER_FLAG_SEEDING);
+ btrfs_set_super_flags(disk_super, super_flags);
+
+ seed_devices->opened = 0;
+ free_fs_devices(seed_devices);
+}
+
/*
* Store the expected generation for seed devices in device items.
*/
@@ -3165,6 +3198,8 @@ error_sysfs:
orig_super_total_bytes);
btrfs_set_super_num_devices(fs_info->super_copy,
orig_super_num_devices);
+ if (seeding_dev)
+ btrfs_rollback_sprout(fs_info, seed_devices);
btrfs_update_per_profile_avail(fs_info);
mutex_unlock(&fs_info->chunk_mutex);
mutex_unlock(&fs_info->fs_devices->device_list_mutex);
diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h
index 0415d74cad9b..337d7007d9e2 100644
--- a/fs/btrfs/volumes.h
+++ b/fs/btrfs/volumes.h
@@ -799,6 +799,7 @@ void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev);
void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev);
void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev,
bool allow_freeze);
+void btrfs_free_device(struct btrfs_device *device);
unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info,
u64 logical);
u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map);
diff --git a/fs/btrfs/zlib.c b/fs/btrfs/zlib.c
index 486b52db583e..e995d0b1452b 100644
--- a/fs/btrfs/zlib.c
+++ b/fs/btrfs/zlib.c
@@ -49,7 +49,7 @@ void zlib_free_workspace(struct list_head *ws)
struct workspace *workspace = list_entry(ws, struct workspace, list);
kvfree(workspace->strm.workspace);
- kfree(workspace->buf);
+ kvfree(workspace->buf);
kfree(workspace);
}
@@ -84,13 +84,13 @@ struct list_head *zlib_alloc_workspace(struct btrfs_fs_info *fs_info, unsigned i
workspace->level = level;
workspace->buf = NULL;
if (need_special_buffer(fs_info)) {
- workspace->buf = kmalloc(ZLIB_DFLTCC_BUF_SIZE,
- __GFP_NOMEMALLOC | __GFP_NORETRY |
- __GFP_NOWARN | GFP_NOIO);
+ workspace->buf = kvmalloc(ZLIB_DFLTCC_BUF_SIZE,
+ __GFP_NOMEMALLOC | __GFP_NORETRY |
+ __GFP_NOWARN | GFP_NOIO);
workspace->buf_size = ZLIB_DFLTCC_BUF_SIZE;
}
if (!workspace->buf) {
- workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL);
+ workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL);
workspace->buf_size = fs_info->sectorsize;
}
if (!workspace->strm.workspace || !workspace->buf)
diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c
index 08a15465a087..c04a9955f2d3 100644
--- a/fs/btrfs/zoned.c
+++ b/fs/btrfs/zoned.c
@@ -123,24 +123,24 @@ static int sb_write_pointer(struct block_device *bdev, struct blk_zone *zones,
} else if (full[0] && full[1]) {
/* Compare two super blocks */
struct address_space *mapping = bdev->bd_mapping;
- struct page *page[BTRFS_NR_SB_LOG_ZONES];
struct btrfs_super_block *super[BTRFS_NR_SB_LOG_ZONES];
for (int i = 0; i < BTRFS_NR_SB_LOG_ZONES; i++) {
u64 zone_end = (zones[i].start + zones[i].capacity) << SECTOR_SHIFT;
u64 bytenr = ALIGN_DOWN(zone_end, BTRFS_SUPER_INFO_SIZE) -
BTRFS_SUPER_INFO_SIZE;
+ struct folio *folio;
filemap_invalidate_lock_shared(mapping);
- page[i] = read_cache_page_gfp(mapping,
- bytenr >> PAGE_SHIFT, GFP_NOFS);
+ folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT,
+ GFP_NOFS);
filemap_invalidate_unlock_shared(mapping);
- if (IS_ERR(page[i])) {
+ if (IS_ERR(folio)) {
if (i == 1)
btrfs_release_disk_super(super[0]);
- return PTR_ERR(page[i]);
+ return PTR_ERR(folio);
}
- super[i] = page_address(page[i]);
+ super[i] = folio_address(folio) + offset_in_folio(folio, bytenr);
}
if (btrfs_super_generation(super[0]) >
@@ -804,15 +804,6 @@ int btrfs_check_mountopts_zoned(const struct btrfs_fs_info *info,
if (!btrfs_is_zoned(info))
return 0;
- /*
- * Space cache writing is not COWed. Disable that to avoid write errors
- * in sequential zones.
- */
- if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) {
- btrfs_err(info, "zoned: space cache v1 is not supported");
- return -EINVAL;
- }
-
if (btrfs_raw_test_opt(*mount_opt, NODATACOW)) {
btrfs_err(info, "zoned: NODATACOW not supported");
return -EINVAL;
diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c
index 58d9ff76fe07..cc92d0b1b948 100644
--- a/fs/btrfs/zstd.c
+++ b/fs/btrfs/zstd.c
@@ -373,7 +373,7 @@ void zstd_free_workspace(struct list_head *ws)
struct workspace *workspace = list_entry(ws, struct workspace, list);
kvfree(workspace->mem);
- kfree(workspace->buf);
+ kvfree(workspace->buf);
kfree(workspace);
}
@@ -391,7 +391,7 @@ struct list_head *zstd_alloc_workspace(struct btrfs_fs_info *fs_info, int level)
workspace->req_level = level;
workspace->last_used = jiffies;
workspace->mem = kvmalloc(workspace->size, GFP_KERNEL | __GFP_NOWARN);
- workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL);
+ workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL);
if (!workspace->mem || !workspace->buf)
goto fail;
@@ -589,10 +589,48 @@ out:
return ret;
}
+/*
+ * Map the destination for the next chunk of output.
+ *
+ * @decompressed is the offset of the next output byte inside the fully
+ * decompressed extent. If that offset has reached the current destination
+ * segment, its page-bounded bio_vec is kmapped so that zstd can write into the
+ * page cache directly, and the number of bytes writable there is returned.
+ * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before
+ * that segment is returned. This covers both the initial prefix and gaps in
+ * the destination bio.
+ */
+static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed,
+ void **kaddr_ret)
+{
+ struct bio *orig_bio = &cb->orig_bbio->bio;
+ struct bio_vec bvec;
+ u32 bvec_offset;
+ u32 off;
+
+ bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter);
+ /*
+ * cb->start may underflow, but subtracting that value can still give us
+ * the correct offset inside the full decompressed extent.
+ */
+ bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start;
+
+ if (decompressed < bvec_offset) {
+ *kaddr_ret = NULL;
+ return bvec_offset - decompressed;
+ }
+
+ off = decompressed - bvec_offset;
+ ASSERT(off < bvec.bv_len);
+ *kaddr_ret = bvec_kmap_local(&bvec) + off;
+ return bvec.bv_len - off;
+}
+
int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
{
struct btrfs_fs_info *fs_info = cb_to_fs_info(cb);
struct workspace *workspace = list_entry(ws, struct workspace, list);
+ struct bio *orig_bio = &cb->orig_bbio->bio;
struct folio_iter fi;
size_t srclen = bio_get_size(&cb->bbio.bio);
zstd_dstream *stream;
@@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
const unsigned int min_folio_size = btrfs_min_folio_size(fs_info);
unsigned long folio_in_index = 0;
unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size);
- unsigned long buf_start;
unsigned long total_out = 0;
bio_first_folio(&fi, &cb->bbio.bio, 0);
@@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
workspace->in_buf.pos = 0;
workspace->in_buf.size = min_t(size_t, srclen, min_folio_size);
- workspace->out_buf.dst = workspace->buf;
- workspace->out_buf.pos = 0;
- workspace->out_buf.size = fs_info->sectorsize;
-
- while (1) {
+ while (orig_bio->bi_iter.bi_size) {
size_t ret2;
+ void *kaddr;
+ u32 dstlen;
+
+ dstlen = zstd_map_dest(cb, total_out, &kaddr);
+ if (kaddr) {
+ workspace->out_buf.dst = kaddr;
+ workspace->out_buf.size = dstlen;
+ } else {
+ workspace->out_buf.dst = workspace->buf;
+ workspace->out_buf.size = min_t(u32, dstlen,
+ fs_info->sectorsize);
+ }
+ workspace->out_buf.pos = 0;
ret2 = zstd_decompress_stream(stream, &workspace->out_buf,
&workspace->in_buf);
+ if (kaddr)
+ kunmap_local(kaddr);
if (unlikely(zstd_is_error(ret2))) {
struct btrfs_inode *inode = cb->bbio.inode;
@@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
ret = -EIO;
goto done;
}
- buf_start = total_out;
total_out += workspace->out_buf.pos;
- workspace->out_buf.pos = 0;
-
- ret = btrfs_decompress_buf2page(workspace->out_buf.dst,
- total_out - buf_start, cb, buf_start);
- if (ret == 0)
- break;
+ if (kaddr)
+ bio_advance(orig_bio, workspace->out_buf.pos);
if (workspace->in_buf.pos >= srclen)
break;