diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-21 10:30:20 +0200 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-21 10:30:20 +0200 |
| commit | ce61374442ff7e350a2c8006357aa6a2c550c08d (patch) | |
| tree | 5dd3c60e8d6bc9bd06e8d391465a1be52f277c1f /fs | |
| parent | 2a945b9b0925f8766ce52986edfb1fb0b8594134 (diff) | |
| parent | 55d05b596923df9445a9f6179b8807d6d4c5a50d (diff) | |
| download | linux-next-ce61374442ff7e350a2c8006357aa6a2c550c08d.tar.gz linux-next-ce61374442ff7e350a2c8006357aa6a2c550c08d.zip | |
Merge branch 'for-next' of ssh://git@gitolite.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git
Diffstat (limited to 'fs')
47 files changed, 1079 insertions, 3030 deletions
diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig index 4b10d78ed99b..0b8d8905e38e 100644 --- a/fs/btrfs/Kconfig +++ b/fs/btrfs/Kconfig @@ -1,4 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 +# misc-next marker config BTRFS_FS tristate "Btrfs filesystem support" diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index cc0bd03048ba..2c6234f6182f 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -27,15 +27,9 @@ struct btrfs_failed_bio { atomic_t repair_count; }; -/* Is this a data path I/O that needs storage layer checksum and repair? */ -static inline bool is_data_bbio(const struct btrfs_bio *bbio) -{ - return bbio->inode && is_data_inode(bbio->inode); -} - static bool bbio_has_ordered_extent(const struct btrfs_bio *bbio) { - return is_data_bbio(bbio) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; + return is_data_inode(bbio->inode) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; } /* @@ -103,7 +97,6 @@ static struct btrfs_bio *btrfs_split_bio(struct btrfs_fs_info *fs_info, bbio->can_use_append = orig_bbio->can_use_append; bbio->is_scrub = orig_bbio->is_scrub; bbio->is_remap = orig_bbio->is_remap; - bbio->async_csum = orig_bbio->async_csum; atomic_inc(&orig_bbio->pending_ios); return bbio; @@ -114,9 +107,6 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status) /* Make sure we're already in task context. */ ASSERT(in_task()); - if (bbio->async_csum) - wait_for_completion(&bbio->csum_done); - bbio->bio.bi_status = status; if (bbio->bio.bi_pool == &btrfs_clone_bioset) { struct btrfs_bio *orig_bbio = bbio->private; @@ -180,30 +170,13 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, struct btrfs_failed_bio *fbio = repair_bbio->private; struct btrfs_inode *inode = repair_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - /* - * We can not move forward the saved_iter, as it will be later - * utilized by repair_bbio again. - */ - struct bvec_iter saved_iter = repair_bbio->saved_iter; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u64 logical = repair_bbio->saved_iter.bi_sector << SECTOR_SHIFT; - const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - unsigned int slot = 0; - /* Repair bbio should be eaxctly one block sized. */ + /* Repair bbio should be exactly one block sized. */ ASSERT(repair_bbio->saved_iter.bi_size == fs_info->sectorsize); - btrfs_bio_for_each_block(paddr, &repair_bbio->bio, &saved_iter, step) { - ASSERT(slot < nr_steps); - paddrs[slot] = paddr; - slot++; - } - if (repair_bbio->bio.bi_status || - !btrfs_data_csum_ok(repair_bbio, dev, 0, paddrs)) { + !btrfs_bio_data_csum_ok(repair_bbio, &repair_bbio->saved_iter, dev)) { bio_reset(&repair_bbio->bio, NULL, REQ_OP_READ); repair_bbio->bio.bi_iter = repair_bbio->saved_iter; @@ -220,9 +193,8 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, do { mirror = prev_repair_mirror(fbio, mirror); - btrfs_repair_io_failure(fs_info, btrfs_ino(inode), - repair_bbio->file_offset, fs_info->sectorsize, - logical, paddrs, step, mirror); + btrfs_repair_bbio_failure(repair_bbio, &repair_bbio->saved_iter, + fs_info->sectorsize, mirror); } while (mirror != fbio->bbio->mirror_num); done: @@ -238,25 +210,21 @@ done: * read succeeded to restore the redundancy. */ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, - u32 bio_offset, - phys_addr_t paddrs[], + const struct bvec_iter *orig_iter, struct btrfs_failed_bio *fbio) { struct btrfs_inode *inode = failed_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - /* - * For bs > ps cases, the saved_iter can be partially moved forward. - * In that case we should round it down to the block boundary. - */ - const u64 logical = round_down(failed_bbio->saved_iter.bi_sector << SECTOR_SHIFT, - sectorsize); struct btrfs_bio *repair_bbio; struct bio *repair_bio; + struct bvec_iter iter = *orig_iter; + const u32 sectorsize = fs_info->sectorsize; + const u32 bio_offset = ((iter.bi_sector - failed_bbio->saved_iter.bi_sector) << + SECTOR_SHIFT); + const u64 logical = (iter.bi_sector << SECTOR_SHIFT); int num_copies; int mirror; + u32 cur = 0; btrfs_debug(fs_info, "repair read error: read error at %llu", failed_bbio->file_offset + bio_offset); @@ -277,17 +245,21 @@ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, atomic_inc(&fbio->repair_count); - repair_bio = bio_alloc_bioset(NULL, nr_steps, REQ_OP_READ, GFP_NOFS, - &btrfs_repair_bioset); + repair_bio = bio_alloc_bioset(NULL, max(1, sectorsize >> PAGE_SHIFT), + REQ_OP_READ, GFP_NOFS, &btrfs_repair_bioset); repair_bio->bi_iter.bi_sector = logical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { + while (cur < sectorsize) { + struct page *page = bio_iter_page(&failed_bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&failed_bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&failed_bbio->bio, iter), + sectorsize - cur); int ret; - ASSERT(offset_in_page(paddrs[i]) + step <= PAGE_SIZE); + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); - ret = bio_add_page(repair_bio, phys_to_page(paddrs[i]), step, - offset_in_page(paddrs[i])); - ASSERT(ret == step); + bio_advance_iter_single(&failed_bbio->bio, &iter, cur_len); + cur += cur_len; } repair_bbio = btrfs_bio(repair_bio); @@ -305,18 +277,16 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - struct bvec_iter *iter = &bbio->saved_iter; + struct bvec_iter iter; blk_status_t status = bbio->bio.bi_status; struct btrfs_failed_bio *fbio = NULL; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - u32 offset = 0; /* Read-repair requires the inode field to be set by the submitter. */ ASSERT(inode); + /* The original bbio should be sectorsize aligned. */ + ASSERT(IS_ALIGNED(bbio->saved_iter.bi_size, sectorsize)); + /* * Hand off repair bios to the repair code as there is no upper level * submitter for them. @@ -329,16 +299,10 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de /* Clear the I/O error. A failed repair will reset it. */ bbio->bio.bi_status = BLK_STS_OK; - btrfs_bio_for_each_block(paddr, &bbio->bio, iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - if (IS_ALIGNED(offset, sectorsize)) { - if (status || - !btrfs_data_csum_ok(bbio, dev, offset - sectorsize, paddrs)) - fbio = repair_one_sector(bbio, offset - sectorsize, - paddrs, fbio); - } + for (iter = bbio->saved_iter; iter.bi_size; + bio_advance_iter(&bbio->bio, &iter, sectorsize)) { + if (status || !btrfs_bio_data_csum_ok(bbio, &iter, dev)) + fbio = repair_one_sector(bbio, &iter, fbio); } if (bbio->csum != bbio->csum_inline) kvfree(bbio->csum); @@ -386,7 +350,7 @@ static void simple_end_io_work(struct work_struct *work) if (bio_op(bio) == REQ_OP_READ) { /* Metadata reads are checked and repaired by the submitter. */ - if (is_data_bbio(bbio)) + if (is_data_inode(bbio->inode)) return btrfs_check_read_bio(bbio, bbio->bio.bi_private); return btrfs_bio_end_io(bbio, bbio->bio.bi_status); } @@ -420,7 +384,7 @@ static void btrfs_raid56_end_io(struct bio *bio) btrfs_bio_counter_dec(bioc->fs_info); bbio->mirror_num = bioc->mirror_num; - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) btrfs_check_read_bio(bbio, NULL); else btrfs_bio_end_io(bbio, bbio->bio.bi_status); @@ -783,7 +747,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * our bio to the physical disk location, so we need to save the * original bytenr so we know what we're checksumming. */ - if (bio_op(bio) == REQ_OP_WRITE && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_WRITE && is_data_inode(bbio->inode)) bbio->orig_logical = logical; bbio->can_use_append = btrfs_use_zone_append(bbio); @@ -809,7 +773,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * Save the iter for the end_io handler and preload the checksums for * data reads. */ - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) { + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) { bbio->saved_iter = bio->bi_iter; ret = btrfs_lookup_bio_sums(bbio); status = errno_to_blk_status(ret); @@ -818,7 +782,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) } if (btrfs_op(bio) == BTRFS_MAP_WRITE) { - if (is_data_bbio(bbio) && bioc && bioc->use_rst) { + if (is_data_inode(bbio->inode) && bioc && bioc->use_rst) { /* * No locking for the list update, as we only add to * the list in the I/O submission path, and list @@ -925,21 +889,23 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * The I/O is issued synchronously to block the repair read completion from * freeing the bio. * - * @ino: Offending inode number - * @fileoff: File offset inside the inode + * @bbio: Original bbio where the repair is needed + * @orig_iter: Points to where the repair starts * @length: Length of the repair write - * @logical: Logical address of the range - * @paddrs: Physical address array of the content - * @step: Length of for each paddrs * @mirror_num: Mirror number to write to. Must not be zero */ -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num) +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num) { - const u32 nr_steps = DIV_ROUND_UP_POW2(length, step); + struct btrfs_inode *inode = bbio->inode; + struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_io_stripe smap = { 0 }; - struct bio *bio = NULL; + struct bvec_iter iter = *orig_iter; + struct bio *repair_bio = NULL; + const u64 logical = iter.bi_sector << SECTOR_SHIFT; + const u64 fileoff = bbio->file_offset + + ((iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT); + u32 cur = 0; int ret = 0; BUG_ON(!mirror_num); @@ -950,8 +916,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, ASSERT(IS_ALIGNED(fileoff, fs_info->sectorsize)); /* Either it's a single data or metadata block. */ ASSERT(length <= BTRFS_MAX_BLOCKSIZE); - ASSERT(step <= length); - ASSERT(is_power_of_2(step)); + + /* Our current iter should not be before the original bbio saved_iter. */ + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); /* * The fs either mounted RO or hit critical errors, no need @@ -979,15 +946,22 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, goto out_counter_dec; } - bio = bio_alloc(smap.dev->bdev, nr_steps, REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); - bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { - ret = bio_add_page(bio, phys_to_page(paddrs[i]), step, offset_in_page(paddrs[i])); - /* We should have allocated enough slots to contain all the different pages. */ - ASSERT(ret == step); + repair_bio = bio_alloc(smap.dev->bdev, max(1, length >> PAGE_SHIFT), + REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); + repair_bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; + while (cur < length) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), length - cur); + + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; } - ret = submit_bio_wait(bio); - bio_put(bio); + + ret = submit_bio_wait(repair_bio); + bio_put(repair_bio); if (ret) { /* try to remap that extent elsewhere? */ btrfs_dev_stat_inc_and_print(smap.dev, BTRFS_DEV_STAT_WRITE_ERRS); @@ -995,8 +969,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, } btrfs_info_rl(fs_info, - "read error corrected: ino %llu off %llu (dev %s sector %llu)", - ino, fileoff, btrfs_dev_name(smap.dev), + "read error corrected: root %llu ino %llu off %llu (dev %s sector %llu)", + btrfs_root_id(inode->root), btrfs_ino(inode), fileoff, + btrfs_dev_name(smap.dev), smap.physical >> SECTOR_SHIFT); ret = 0; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index 303ed6c7103d..bbf362b8668b 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -58,7 +58,6 @@ struct btrfs_bio { struct btrfs_ordered_extent *ordered; struct btrfs_ordered_sum *sums; struct work_struct csum_work; - struct completion csum_done; struct bvec_iter csum_saved_iter; u64 orig_physical; u64 orig_logical; @@ -93,9 +92,6 @@ struct btrfs_bio { /* Whether the bio is coming from copy_remapped_data_io(). */ bool is_remap:1; - /* Whether the csum generation for data write is async. */ - bool async_csum:1; - /* Whether the bio is written using zone append. */ bool can_use_append:1; @@ -126,8 +122,7 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status); void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num); void btrfs_submit_repair_write(struct btrfs_bio *bbio, int mirror_num, bool dev_replace); -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num); +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num); #endif diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index ee182369254c..e6080cb47c89 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -904,22 +904,6 @@ static noinline void caching_thread(struct btrfs_work *work) down_read(&fs_info->commit_root_sem); load_block_group_size_class(caching_ctl); - if (btrfs_test_opt(fs_info, SPACE_CACHE)) { - ret = load_free_space_cache(block_group); - if (ret == 1) { - ret = 0; - goto done; - } - - /* - * We failed to load the space cache, set ourselves to - * CACHE_STARTED and carry on. - */ - spin_lock(&block_group->lock); - block_group->cached = BTRFS_CACHE_STARTED; - spin_unlock(&block_group->lock); - wake_up(&caching_ctl->wait); - } /* * If we are in the transaction that populated the free space tree we @@ -933,7 +917,7 @@ static noinline void caching_thread(struct btrfs_work *work) ret = btrfs_load_free_space_tree(caching_ctl); else ret = load_extent_tree_free(caching_ctl); -done: + spin_lock(&block_group->lock); block_group->caching_ctl = NULL; block_group->cached = ret ? BTRFS_CACHE_ERROR : BTRFS_CACHE_FINISHED; @@ -1194,36 +1178,21 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, goto out; } - /* - * get the inode first so any iput calls done for the io_list - * aren't the final iput (no unlinks allowed now) - */ inode = lookup_free_space_inode(block_group, path); - mutex_lock(&trans->transaction->cache_write_mutex); /* - * Make sure our free space cache IO is done before removing the - * free space inode + * Do not delete the block group item while + * btrfs_start_dirty_block_groups() is updating it. */ + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); spin_lock(&trans->transaction->dirty_bgs_lock); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - WARN_ON(!IS_ERR(inode) && inode != block_group->io_ctl.inode); - - spin_unlock(&trans->transaction->dirty_bgs_lock); - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - spin_lock(&trans->transaction->dirty_bgs_lock); - } - if (!list_empty(&block_group->dirty_list)) { list_del_init(&block_group->dirty_list); remove_rsv = true; btrfs_put_block_group(block_group); } spin_unlock(&trans->transaction->dirty_bgs_lock); - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); ret = btrfs_remove_free_space_inode(trans, inode, block_group); if (unlikely(ret)) { @@ -1287,7 +1256,6 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, spin_lock(&trans->transaction->dirty_bgs_lock); WARN_ON(!list_empty(&block_group->dirty_list)); - WARN_ON(!list_empty(&block_group->io_list)); spin_unlock(&trans->transaction->dirty_bgs_lock); btrfs_remove_free_space_cache(block_group); @@ -1612,7 +1580,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) space_info = block_group->space_info; - if (ret || btrfs_mixed_space_info(space_info)) { + if (btrfs_mixed_space_info(space_info)) { btrfs_put_block_group(block_group); continue; } @@ -1727,6 +1695,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) ret = inc_block_group_ro(block_group, false); up_write(&space_info->groups_sem); if (ret < 0) { + btrfs_link_bg_list(block_group, &retry_list); ret = 0; goto next; } @@ -1749,6 +1718,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) block_group->start); if (IS_ERR(trans)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); ret = PTR_ERR(trans); goto next; } @@ -1759,6 +1729,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) */ if (!clean_pinned_extents(trans, block_group)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); goto end_trans; } @@ -1845,6 +1816,8 @@ end_trans: next: btrfs_put_block_group(block_group); spin_lock(&fs_info->unused_bgs_lock); + if (ret) + break; } list_splice_tail(&retry_list, &fs_info->unused_bgs); spin_unlock(&fs_info->unused_bgs_lock); @@ -2431,7 +2404,6 @@ static struct btrfs_block_group *btrfs_create_block_group( INIT_LIST_HEAD(&cache->ro_list); INIT_LIST_HEAD(&cache->discard_list); INIT_LIST_HEAD(&cache->dirty_list); - INIT_LIST_HEAD(&cache->io_list); INIT_LIST_HEAD(&cache->active_bg_list); btrfs_init_free_space_ctl(cache, cache->free_space_ctl); atomic_set(&cache->frozen, 0); @@ -2487,8 +2459,7 @@ static int check_chunk_block_group_mappings(struct btrfs_fs_info *fs_info) static int read_one_block_group(struct btrfs_fs_info *info, struct btrfs_block_group_item_v2 *bgi, - const struct btrfs_key *key, - bool need_clear) + const struct btrfs_key *key) { struct btrfs_block_group *cache; const bool mixed = btrfs_fs_incompat(info, MIXED_GROUPS); @@ -2514,20 +2485,6 @@ static int read_one_block_group(struct btrfs_fs_info *info, btrfs_set_free_space_tree_thresholds(cache); - if (need_clear) { - /* - * When we mount with old space cache, we need to - * set BTRFS_DC_CLEAR and set dirty flag. - * - * a) Setting 'BTRFS_DC_CLEAR' makes sure that we - * truncate the old free space cache inode and - * setup a new one. - * b) Setting 'dirty flag' makes sure that we flush - * the new space cache info onto disk. - */ - if (btrfs_test_opt(info, SPACE_CACHE)) - cache->disk_cache_state = BTRFS_DC_CLEAR; - } if (!mixed && ((cache->flags & BTRFS_BLOCK_GROUP_METADATA) && (cache->flags & BTRFS_BLOCK_GROUP_DATA))) { btrfs_err(info, @@ -2668,8 +2625,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) struct btrfs_block_group *cache; struct btrfs_space_info *space_info; struct btrfs_key key; - bool need_clear = false; - u64 cache_gen; /* * Either no extent root (with ibadroots rescue option) or we have @@ -2690,13 +2645,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) if (!path) return -ENOMEM; - cache_gen = btrfs_super_cache_generation(info->super_copy); - if (btrfs_test_opt(info, SPACE_CACHE) && - btrfs_super_generation(info->super_copy) != cache_gen) - need_clear = true; - if (btrfs_test_opt(info, CLEAR_CACHE)) - need_clear = true; - while (1) { struct btrfs_block_group_item_v2 bgi; struct extent_buffer *leaf; @@ -2725,7 +2673,7 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) btrfs_item_key_to_cpu(leaf, &key, slot); btrfs_release_path(path); - ret = read_one_block_group(info, &bgi, &key, need_clear); + ret = read_one_block_group(info, &bgi, &key); if (ret < 0) goto error; key.objectid += key.offset; @@ -3373,207 +3321,16 @@ fail: } -static void cache_save_setup(struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct inode *inode = NULL; - struct extent_changeset *data_reserved = NULL; - u64 alloc_hint = 0; - int dcs = BTRFS_DC_ERROR; - u64 cache_size = 0; - int retries = 0; - int ret = 0; - - if (!btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - - /* - * If this block group is smaller than 100 megs don't bother caching the - * block group. - */ - if (block_group->length < (100 * SZ_1M)) { - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - return; - } - - if (TRANS_ABORTED(trans)) - return; -again: - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode) && PTR_ERR(inode) != -ENOENT) { - ret = PTR_ERR(inode); - btrfs_release_path(path); - goto out; - } - - if (IS_ERR(inode)) { - if (retries) { - ret = PTR_ERR(inode); - btrfs_err(fs_info, - "failed to lookup free space inode after creation for block group %llu: %d", - block_group->start, ret); - goto out_free; - } - retries++; - - if (block_group->ro) - goto out_free; - - ret = create_free_space_inode(trans, block_group, path); - if (ret) - goto out_free; - goto again; - } - - /* - * We want to set the generation to 0, that way if anything goes wrong - * from here on out we know not to trust this cache when we load up next - * time. - */ - BTRFS_I(inode)->generation = 0; - ret = btrfs_update_inode(trans, BTRFS_I(inode)); - if (unlikely(ret)) { - /* - * So theoretically we could recover from this, simply set the - * super cache generation to 0 so we know to invalidate the - * cache, but then we'd have to keep track of the block groups - * that fail this way so we know we _have_ to reset this cache - * before the next commit or risk reading stale cache. So to - * limit our exposure to horrible edge cases lets just abort the - * transaction, this only happens in really bad situations - * anyway. - */ - btrfs_abort_transaction(trans, ret); - goto out_put; - } - - /* We've already setup this transaction, go ahead and exit */ - if (block_group->cache_generation == trans->transid && - i_size_read(inode)) { - dcs = BTRFS_DC_SETUP; - goto out_put; - } - - if (i_size_read(inode) > 0) { - ret = btrfs_check_trunc_cache_free_space(fs_info, - &fs_info->global_block_rsv); - if (ret) - goto out_put; - - ret = btrfs_truncate_free_space_cache(trans, NULL, inode); - if (ret) - goto out_put; - } - - spin_lock(&block_group->lock); - if (block_group->cached != BTRFS_CACHE_FINISHED || - !btrfs_test_opt(fs_info, SPACE_CACHE)) { - /* - * don't bother trying to write stuff out _if_ - * a) we're not cached, - * b) we're with nospace_cache mount option, - * c) we're with v2 space_cache (FREE_SPACE_TREE). - */ - dcs = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - goto out_put; - } - spin_unlock(&block_group->lock); - - /* - * We hit an ENOSPC when setting up the cache in this transaction, just - * skip doing the setup, we've already cleared the cache so we're safe. - */ - if (test_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags)) - goto out_put; - - /* - * Try to preallocate enough space based on how big the block group is. - * Keep in mind this has to include any pinned space which could end up - * taking up quite a bit since it's not folded into the other space - * cache. - */ - cache_size = div_u64(block_group->length, SZ_256M); - if (!cache_size) - cache_size = 1; - - cache_size *= 16; - cache_size *= fs_info->sectorsize; - - ret = btrfs_check_data_free_space(BTRFS_I(inode), &data_reserved, 0, - cache_size, false); - if (ret) - goto out_put; - - ret = btrfs_prealloc_file_range_trans(inode, trans, 0, 0, cache_size, - cache_size, cache_size, - &alloc_hint); - /* - * Our cache requires contiguous chunks so that we don't modify a bunch - * of metadata or split extents when writing the cache out, which means - * we can enospc if we are heavily fragmented in addition to just normal - * out of space conditions. So if we hit this just skip setting up any - * other block groups for this transaction, maybe we'll unpin enough - * space the next time around. - */ - if (!ret) - dcs = BTRFS_DC_SETUP; - else if (ret == -ENOSPC) - set_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags); - -out_put: - iput(inode); -out_free: - btrfs_release_path(path); -out: - spin_lock(&block_group->lock); - if (!ret && dcs == BTRFS_DC_SETUP) - block_group->cache_generation = trans->transid; - block_group->disk_cache_state = dcs; - spin_unlock(&block_group->lock); - - extent_changeset_free(data_reserved); -} - -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct btrfs_block_group *cache, *tmp; - struct btrfs_transaction *cur_trans = trans->transaction; - BTRFS_PATH_AUTO_FREE(path); - - if (list_empty(&cur_trans->dirty_bgs) || - !btrfs_test_opt(fs_info, SPACE_CACHE)) - return 0; - - path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; - - /* Could add new block groups, use _safe just in case */ - list_for_each_entry_safe(cache, tmp, &cur_trans->dirty_bgs, - dirty_list) { - if (cache->disk_cache_state == BTRFS_DC_CLEAR) - cache_save_setup(cache, trans, path); - } - - return 0; -} - /* - * Transaction commit does final block group cache writeback during a critical + * Transaction commit does the final block group item updates during a critical * section where nothing is allowed to change the FS. This is required in - * order for the cache to actually match the block group, but can introduce a + * order for the items to actually match the block groups, but can introduce a * lot of latency into the commit. * - * So, btrfs_start_dirty_block_groups is here to kick off block group cache IO. - * There's a chance we'll have to redo some of it if the block group changes - * again during the commit, but it greatly reduces the commit latency by - * getting rid of the easy block groups while we're still allowing others to + * So, btrfs_start_dirty_block_groups is here to update the block group items + * early. There's a chance we'll have to redo some of it if the block group + * changes again during the commit, but it greatly reduces the commit latency + * by getting rid of the easy block groups while we're still allowing others to * join the commit. */ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) @@ -3582,10 +3339,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); LIST_HEAD(dirty); - struct list_head *io = &cur_trans->io_bgs; int loops = 0; spin_lock(&cur_trans->dirty_bgs_lock); @@ -3609,33 +3364,18 @@ again: } /* - * cache_write_mutex is here only to save us from balance or automatic - * removal of empty block groups deleting this block group while we are - * writing out the cache + * dirty_bgs_update_mutex is here only to save us from balance or + * automatic removal of empty block groups deleting this block group + * while we are updating its item */ - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); while (!list_empty(&dirty)) { bool drop_reserve = true; cache = list_first_entry(&dirty, struct btrfs_block_group, dirty_list); - /* - * This can happen if something re-dirties a block group that - * is already under IO. Just wait for it to finish and then do - * it all again - */ - if (!list_empty(&cache->io_list)) { - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - /* - * btrfs_wait_cache_io uses the cache->dirty_list to decide if - * it should update the cache_state. Don't delete until after - * we wait. - * * Since we're not running in the commit critical section * we need the dirty_bgs_lock to protect from update_block_group */ @@ -3643,72 +3383,39 @@ again: list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); - - if (cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - - /* - * The cache_write_mutex is protecting the - * io_list, also refer to the definition of - * btrfs_transaction::io_bgs for more details - */ - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } - if (!ret) { - ret = update_block_group_item(trans, path, cache); - /* - * Our block group might still be attached to the list - * of new block groups in the transaction handle of some - * other task (struct btrfs_trans_handle->new_bgs). This - * means its block group item isn't yet in the extent - * tree. If this happens ignore the error, as we will - * try again later in the critical section of the - * transaction commit. - */ - if (ret == -ENOENT) { - ret = 0; - spin_lock(&cur_trans->dirty_bgs_lock); - if (list_empty(&cache->dirty_list)) { - list_add_tail(&cache->dirty_list, - &cur_trans->dirty_bgs); - btrfs_get_block_group(cache); - drop_reserve = false; - } - spin_unlock(&cur_trans->dirty_bgs_lock); - } else if (ret) { - btrfs_abort_transaction(trans, ret); + ret = update_block_group_item(trans, path, cache); + /* + * Our block group might still be attached to the list of new + * block groups in the transaction handle of some other task + * (struct btrfs_trans_handle->new_bgs). This means its block + * group item isn't yet in the extent tree. If this happens + * ignore the error, as we will try again later in the critical + * section of the transaction commit. + */ + if (ret == -ENOENT) { + ret = 0; + spin_lock(&cur_trans->dirty_bgs_lock); + if (list_empty(&cache->dirty_list)) { + list_add_tail(&cache->dirty_list, + &cur_trans->dirty_bgs); + btrfs_get_block_group(cache); + drop_reserve = false; } + spin_unlock(&cur_trans->dirty_bgs_lock); + } else if (ret) { + btrfs_abort_transaction(trans, ret); } - /* If it's not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); if (drop_reserve) btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); - /* - * Avoid blocking other tasks for too long. It might even save - * us from writing caches for block groups that are going to be - * removed. - */ - mutex_unlock(&trans->transaction->cache_write_mutex); + /* Avoid blocking other tasks for too long. */ + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); if (ret) goto out; - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); } - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); /* * Go through delayed refs for all the stuff we've just kicked off @@ -3722,7 +3429,7 @@ again: list_splice_init(&cur_trans->dirty_bgs, &dirty); /* * dirty_bgs_lock protects us from concurrent block group - * deletes too (not just cache_write_mutex). + * deletes too (not just dirty_bgs_update_mutex). */ if (!list_empty(&dirty)) { spin_unlock(&cur_trans->dirty_bgs_lock); @@ -3747,121 +3454,34 @@ int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); - struct list_head *io = &cur_trans->io_bgs; path = btrfs_alloc_path(); if (!path) return -ENOMEM; - /* - * Even though we are in the critical section of the transaction commit, - * we can still have concurrent tasks adding elements to this - * transaction's list of dirty block groups. These tasks correspond to - * endio free space workers started when writeback finishes for a - * space cache, which run inode.c:btrfs_finish_ordered_io(), and can - * allocate new block groups as a result of COWing nodes of the root - * tree when updating the free space inode. The writeback for the space - * caches is triggered by an earlier call to - * btrfs_start_dirty_block_groups() and iterations of the following - * loop. - * Also we want to do the cache_save_setup first and then run the - * delayed refs to make sure we have the best chance at doing this all - * in one shot. - */ spin_lock(&cur_trans->dirty_bgs_lock); while (!list_empty(&cur_trans->dirty_bgs)) { cache = list_first_entry(&cur_trans->dirty_bgs, struct btrfs_block_group, dirty_list); - - /* - * This can happen if cache_save_setup re-dirties a block group - * that is already under IO. Just wait for it to finish and - * then do it all again - */ - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - - /* - * Don't remove from the dirty list until after we've waited on - * any pending IO - */ list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); if (!ret) ret = btrfs_run_delayed_refs(trans, U64_MAX); - - if (!ret && cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } if (!ret) { ret = update_block_group_item(trans, path, cache); - /* - * One of the free space endio workers might have - * created a new block group while updating a free space - * cache's inode (at inode.c:btrfs_finish_ordered_io()) - * and hasn't released its transaction handle yet, in - * which case the new block group is still attached to - * its transaction handle and its creation has not - * finished yet (no block group item in the extent tree - * yet, etc). If this is the case, wait for all free - * space endio workers to finish and retry. This is a - * very rare case so no need for a more efficient and - * complex approach. - */ - if (ret == -ENOENT) { - wait_event(cur_trans->writer_wait, - atomic_read(&cur_trans->num_writers) == 1); - ret = update_block_group_item(trans, path, cache); - if (ret) - btrfs_abort_transaction(trans, ret); - } else if (ret) { + if (ret) btrfs_abort_transaction(trans, ret); - } } - /* If its not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(io)) { - cache = list_first_entry(io, struct btrfs_block_group, - io_list); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - return ret; } @@ -3905,10 +3525,10 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, factor = btrfs_bg_type_to_factor(cache->flags); /* - * If this block group has free space cache written out, we need to make - * sure to load it if we are removing space. This is because we need - * the unpinning stage to actually add the space back to the block group, - * otherwise we will leak space. + * Make sure the free space of this block group is loaded if we are + * removing space. This is because we need the unpinning stage to + * actually add the space back to the block group, otherwise we will + * leak space. */ if (!alloc && !btrfs_block_group_done(cache)) btrfs_cache_block_group(cache, true); @@ -3916,10 +3536,6 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, spin_lock(&space_info->lock); spin_lock(&cache->lock); - if (btrfs_test_opt(info, SPACE_CACHE) && - cache->disk_cache_state < BTRFS_DC_CLEAR) - cache->disk_cache_state = BTRFS_DC_CLEAR; - old_val = cache->used; if (alloc) { old_val += num_bytes; @@ -3964,7 +3580,7 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, /* * No longer have used bytes in this block group, queue it for deletion. * We do this after adding the block group to the dirty list to avoid - * races between cleaner kthread and space cache writeout. + * races between the cleaner kthread and the dirty block group writeout. */ if (!alloc && old_val == 0) { if (!btrfs_test_opt(info, DISCARD_ASYNC)) @@ -4640,7 +4256,6 @@ void btrfs_put_block_group_cache(struct btrfs_fs_info *info) block_group->inode = NULL; spin_unlock(&block_group->lock); - ASSERT(block_group->io_ctl.inode == NULL); iput(&inode->vfs_inode); } else { spin_unlock(&block_group->lock); @@ -4777,7 +4392,6 @@ int btrfs_free_block_groups(struct btrfs_fs_info *info) btrfs_remove_free_space_cache(block_group); ASSERT(block_group->cached != BTRFS_CACHE_STARTED); ASSERT(list_empty(&block_group->dirty_list)); - ASSERT(list_empty(&block_group->io_list)); ASSERT(list_empty(&block_group->bg_list)); ASSERT(refcount_read(&block_group->refs) == 1); ASSERT(block_group->swap_extents == 0); diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index 69d56864d4ba..d567ed822e55 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -20,13 +20,6 @@ struct btrfs_fs_info; struct btrfs_inode; struct btrfs_trans_handle; -enum btrfs_disk_cache_state { - BTRFS_DC_WRITTEN, - BTRFS_DC_ERROR, - BTRFS_DC_CLEAR, - BTRFS_DC_SETUP, -}; - enum btrfs_block_group_size_class { /* Unset */ BTRFS_BG_SZ_NONE, @@ -131,11 +124,10 @@ struct btrfs_block_group { u64 delalloc_bytes; u64 bytes_super; u64 flags; - u64 cache_generation; u64 global_root_id; u64 remap_bytes; u32 identity_remap_count; - /* The last commited identity_remap_count value of this block group. */ + /* The last committed identity_remap_count value of this block group. */ u32 last_identity_remap_count; /* * The last committed used bytes of this block group, if the above @used @@ -171,8 +163,6 @@ struct btrfs_block_group { unsigned long full_stripe_len; unsigned long runtime_flags; - enum btrfs_disk_cache_state disk_cache_state; - /* Cache tracking stuff */ enum btrfs_caching_type cached; struct btrfs_caching_control *caching_ctl; @@ -228,9 +218,6 @@ struct btrfs_block_group { /* For dirty block groups */ struct list_head dirty_list; - struct list_head io_list; - - struct btrfs_io_ctl io_ctl; /* * Incremented when doing extent allocations and holding a read lock @@ -368,7 +355,6 @@ int btrfs_inc_block_group_ro(struct btrfs_block_group *cache, void btrfs_dec_block_group_ro(struct btrfs_block_group *cache); int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans); int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans); -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans); int btrfs_update_block_group(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, bool alloc); int btrfs_add_reserved_bytes(struct btrfs_block_group *cache, diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 1082fa92c145..114a5c38afd3 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -32,6 +32,7 @@ struct btrfs_trans_handle; struct btrfs_bio; struct btrfs_file_extent; struct btrfs_delayed_node; +struct btrfs_dir_index_prealloc; /* * Since we search a directory based on f_pos (struct dir_context::pos) we have @@ -402,8 +403,6 @@ static inline void btrfs_mod_outstanding_extents(struct btrfs_inode *inode, { lockdep_assert_held(&inode->lock); inode->outstanding_extents += mod; - if (btrfs_is_free_space_inode(inode)) - return; trace_btrfs_inode_mod_outstanding_extents(inode->root, btrfs_ino(inode), mod, inode->outstanding_extents); } @@ -507,14 +506,10 @@ static inline void btrfs_set_inode_mapping_order(struct btrfs_inode *inode) inode->root->fs_info->block_max_order); } -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest); -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest); -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected); -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]); +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + struct btrfs_device *dev); +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); @@ -527,7 +522,8 @@ int btrfs_unlink_inode(struct btrfs_trans_handle *trans, const struct fscrypt_str *name); int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index); + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc); int btrfs_delete_subvolume(struct btrfs_inode *dir, struct dentry *dentry); int btrfs_truncate_block(struct btrfs_inode *inode, u64 offset, u64 start, u64 end); @@ -594,10 +590,6 @@ int btrfs_wait_on_delayed_iputs(struct btrfs_fs_info *fs_info); int btrfs_prealloc_file_range(struct inode *inode, int mode, u64 start, u64 num_bytes, u64 min_size, loff_t actual_len, u64 *alloc_hint); -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint); int btrfs_run_delalloc_range(struct btrfs_inode *inode, struct folio *locked_folio, u64 start, u64 end, struct writeback_control *wbc); void btrfs_queue_writepage_fixup(struct btrfs_inode *inode, struct folio *folio); diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index c62b5148d5ac..ad90e14032db 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -168,10 +168,10 @@ static unsigned long btrfs_compr_pool_scan(struct shrinker *sh, struct shrink_co spin_unlock(&compr_pool.lock); list_for_each_safe(tmp, next, &remove) { - struct page *page = list_entry(tmp, struct page, lru); + struct folio *folio = list_entry(tmp, struct folio, lru); - ASSERT(page_ref_count(page) == 1); - put_page(page); + ASSERT(folio_ref_count(folio) == 1); + folio_put(folio); } return freed; @@ -431,7 +431,7 @@ static noinline int add_ra_bio_folios(struct inode *inode, u64 compressed_end, } /* - * Since add_ra_bio_pages() is always speculative, suppress + * Since add_ra_bio_folios() is always speculative, suppress * allocation warnings. */ masked_constraint_gfp = mapping_gfp_constraint(mapping, constraint_gfp); @@ -960,7 +960,7 @@ bool btrfs_compress_level_valid(unsigned int type, int level) return levels->min_level <= level && level <= levels->max_level; } -/* Wrapper around find_get_page(), with extra error message. */ +/* Wrapper around filemap_get_folio(), with extra error message. */ int btrfs_compress_filemap_get_folio(struct address_space *mapping, u64 start, struct folio **in_folio_ret) { @@ -1488,10 +1488,11 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws) static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { - struct page *page; - pgoff_t index, index_end; - u32 i, curr_sample_pos; - u8 *in_data; + const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize; + u64 cur = start; + u32 curr_sample_pos = 0; + + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); /* * Compression handles the input data by chunks of 128KiB @@ -1502,38 +1503,30 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, * MAX_SAMPLE_SIZE - calculated under assumption that heuristic will * process no more than BTRFS_MAX_UNCOMPRESSED at a time. */ - if (end - start > BTRFS_MAX_UNCOMPRESSED) - end = start + BTRFS_MAX_UNCOMPRESSED; - - index = start >> PAGE_SHIFT; - index_end = end >> PAGE_SHIFT; - - /* Don't miss unaligned end */ - if (!PAGE_ALIGNED(end)) - index_end++; - - curr_sample_pos = 0; - while (index < index_end) { - page = find_get_page(inode->i_mapping, index); - in_data = kmap_local_page(page); - /* Handle case where the start is not aligned to PAGE_SIZE */ - i = start % PAGE_SIZE; - while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { - /* Don't sample any garbage from the last page */ - if (start > end - SAMPLING_READ_SIZE) - break; - memcpy(&ws->sample[curr_sample_pos], &in_data[i], - SAMPLING_READ_SIZE); - i += SAMPLING_INTERVAL; - start += SAMPLING_INTERVAL; + if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED) + end = start + BTRFS_MAX_UNCOMPRESSED - 1; + + while (cur < end) { + struct folio *folio; + void *in_data; + u64 next_pos; + + folio = filemap_get_folio(inode->i_mapping, cur >> PAGE_SHIFT); + /* All folios inside the range should exist and be locked. */ + ASSERT(!IS_ERR(folio)); + next_pos = min_t(u64, end + 1, folio_next_pos(folio)); + in_data = kmap_local_folio(folio, 0); + + for (; cur < next_pos; cur += SAMPLING_INTERVAL) { + memcpy(&ws->sample[curr_sample_pos], + in_data + offset_in_folio(folio, cur), + SAMPLING_READ_SIZE); curr_sample_pos += SAMPLING_READ_SIZE; } kunmap_local(in_data); - put_page(page); - - index++; + folio_put(folio); + cur = next_pos; } - ws->sample_size = curr_sample_pos; } diff --git a/fs/btrfs/ctree.c b/fs/btrfs/ctree.c index 8fe330d81b8f..fef0e49dd918 100644 --- a/fs/btrfs/ctree.c +++ b/fs/btrfs/ctree.c @@ -3943,7 +3943,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, orig_offset = btrfs_item_offset(leaf, path->slots[0]); item_size = btrfs_item_size(leaf, path->slots[0]); - buf = kmalloc(item_size, GFP_NOFS); + buf = kvmalloc(item_size, GFP_NOFS); if (!buf) return -ENOMEM; @@ -3981,7 +3981,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, btrfs_mark_buffer_dirty(trans, leaf); BUG_ON(btrfs_leaf_free_space(leaf) < 0); - kfree(buf); + kvfree(buf); return 0; } diff --git a/fs/btrfs/delalloc-space.c b/fs/btrfs/delalloc-space.c index d357ed7efd99..77781852e417 100644 --- a/fs/btrfs/delalloc-space.c +++ b/fs/btrfs/delalloc-space.c @@ -132,9 +132,7 @@ int btrfs_alloc_data_chunk_ondemand(const struct btrfs_inode *inode, u64 bytes) /* Make sure bytes are sectorsize aligned */ bytes = ALIGN(bytes, fs_info->sectorsize); - if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; - else if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) + if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) flush = BTRFS_RESERVE_FLUSH_ZONED_RELOCATION; return btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), bytes, flush); @@ -155,8 +153,6 @@ int btrfs_check_data_free_space(struct btrfs_inode *inode, if (noflush) flush = BTRFS_RESERVE_NO_FLUSH; - else if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; ret = btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), len, flush); if (ret < 0) @@ -326,15 +322,10 @@ int btrfs_delalloc_reserve_metadata(struct btrfs_inode *inode, u64 num_bytes, int ret = 0; /* - * If we are a free space inode we need to not flush since we will be in - * the middle of a transaction commit. We also don't need the delalloc - * mutex since we won't race with anybody. We need this mostly to make - * lockdep shut its filthy mouth. - * * If we have a transaction open (can happen if we call truncate_block * from truncate), then we need FLUSH_LIMIT so we don't deadlock. */ - if (noflush || btrfs_is_free_space_inode(inode)) { + if (noflush) { flush = BTRFS_RESERVE_NO_FLUSH; } else { if (current->journal_info) diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c index db2ffab0941a..1f20148c1fd4 100644 --- a/fs/btrfs/delayed-inode.c +++ b/fs/btrfs/delayed-inode.c @@ -6,6 +6,7 @@ #include <linux/slab.h> #include <linux/iversion.h> +#include <linux/error-injection.h> #include "ctree.h" #include "fs.h" #include "messages.h" @@ -686,7 +687,7 @@ static int btrfs_insert_delayed_item(struct btrfs_trans_handle *trans, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(first_item->bytes_reserved == 0); @@ -1469,35 +1470,93 @@ static void btrfs_release_dir_index_item_space(struct btrfs_trans_handle *trans) trans->bytes_reserved -= bytes; } -/* Will return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index) +/* + * Pre-allocate a delayed node and delayed item for a dir index insertion and + * copy the name into the item. Call this before modifying the btree so that + * ENOMEM can be returned before any on-disk state has changed. + * + * The returned prealloc is consumed by either + * btrfs_insert_delayed_dir_index_prealloc() or + * btrfs_free_delayed_dir_index_prealloc(); it must not be used afterwards. + * + * Returns a prealloc on success, ERR_PTR on allocation failure. + */ +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len) +{ + struct btrfs_dir_index_prealloc *prealloc; + struct btrfs_delayed_node *node; + struct btrfs_delayed_item *item; + + prealloc = kzalloc_obj(*prealloc, GFP_NOFS); + if (!prealloc) + return ERR_PTR(-ENOMEM); + + node = btrfs_get_or_create_delayed_node(dir, &prealloc->tracker); + if (IS_ERR(node)) { + kfree(prealloc); + return ERR_CAST(node); + } + + item = btrfs_alloc_delayed_item(sizeof(struct btrfs_dir_item) + name_len, + node, BTRFS_DELAYED_INSERTION_ITEM); + if (!item) { + btrfs_release_delayed_node(node, &prealloc->tracker); + kfree(prealloc); + return ERR_PTR(-ENOMEM); + } + + memcpy(item->data + sizeof(struct btrfs_dir_item), name, name_len); + + prealloc->node = node; + prealloc->item = item; + return prealloc; +} +ALLOW_ERROR_INJECTION(btrfs_prealloc_delayed_dir_index, ERRNO); + +/* + * Free resources from btrfs_prealloc_delayed_dir_index() when the btree + * insertion failed and we will not commit the delayed dir index. Does nothing + * if @prealloc is NULL. + */ +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc) { + if (!prealloc) + return; + + btrfs_release_delayed_item(prealloc->item); + btrfs_release_dir_index_item_space(trans); + btrfs_release_delayed_node(prealloc->node, &prealloc->tracker); + kfree(prealloc); +} + +/* + * Commit a pre-allocated delayed dir index item. @prealloc must have been + * returned by btrfs_prealloc_delayed_dir_index(). This populates the item, + * adds it to the delayed node's rb-tree, and reserves metadata space. It + * cannot fail with ENOMEM. @prealloc is freed here in all cases. + * + * Return 0 or -EEXIST (index number collision, unexpected). + */ +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index) +{ + struct btrfs_delayed_node *delayed_node = prealloc->node; + struct btrfs_ref_tracker *tracker = &prealloc->tracker; + struct btrfs_delayed_item *delayed_item = prealloc->item; struct btrfs_fs_info *fs_info = trans->fs_info; const unsigned int leaf_data_size = BTRFS_LEAF_DATA_SIZE(fs_info); - struct btrfs_delayed_node *delayed_node; - struct btrfs_ref_tracker delayed_node_tracker; - struct btrfs_delayed_item *delayed_item; + const int name_len = delayed_item->data_len - sizeof(struct btrfs_dir_item); struct btrfs_dir_item *dir_item; bool reserve_leaf_space; u32 data_len; int ret; - delayed_node = btrfs_get_or_create_delayed_node(dir, &delayed_node_tracker); - if (IS_ERR(delayed_node)) - return PTR_ERR(delayed_node); - - delayed_item = btrfs_alloc_delayed_item(sizeof(*dir_item) + name_len, - delayed_node, - BTRFS_DELAYED_INSERTION_ITEM); - if (!delayed_item) { - ret = -ENOMEM; - goto release_node; - } - delayed_item->index = index; dir_item = (struct btrfs_dir_item *)delayed_item->data; @@ -1506,7 +1565,7 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, btrfs_set_stack_dir_data_len(dir_item, 0); btrfs_set_stack_dir_name_len(dir_item, name_len); btrfs_set_stack_dir_flags(dir_item, flags); - memcpy((char *)(dir_item + 1), name, name_len); + /* Name was already copied by btrfs_prealloc_delayed_dir_index(). */ data_len = delayed_item->data_len + sizeof(struct btrfs_item); @@ -1524,7 +1583,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, if (unlikely(ret)) { btrfs_err(trans->fs_info, "error adding delayed dir index item, name: %.*s, index: %llu, root: %llu, dir: %llu, dir->index_cnt: %llu, delayed_node->index_cnt: %llu, error: %pe", - name_len, name, index, btrfs_root_id(delayed_node->root), + name_len, + (const char *)(dir_item + 1), + index, btrfs_root_id(delayed_node->root), delayed_node->inode_id, dir->index_cnt, delayed_node->index_cnt, ERR_PTR(ret)); btrfs_release_delayed_item(delayed_item); @@ -1562,7 +1623,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, mutex_unlock(&delayed_node->mutex); release_node: - btrfs_release_delayed_node(delayed_node, &delayed_node_tracker); + /* Must release the node before freeing @tracker's containing struct. */ + btrfs_release_delayed_node(delayed_node, tracker); + kfree(prealloc); return ret; } @@ -1581,7 +1644,7 @@ static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(item->bytes_reserved == 0); diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h index fc752863f89b..57ba96cfaf9c 100644 --- a/fs/btrfs/delayed-inode.h +++ b/fs/btrfs/delayed-inode.h @@ -115,11 +115,23 @@ struct btrfs_delayed_item { }; void btrfs_init_delayed_root(struct btrfs_delayed_root *delayed_root); -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index); + +struct btrfs_dir_index_prealloc { + struct btrfs_delayed_node *node; + struct btrfs_ref_tracker tracker; + struct btrfs_delayed_item *item; +}; + +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len); +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc); +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index); int btrfs_delete_delayed_dir_index(struct btrfs_trans_handle *trans, struct btrfs_inode *dir, u64 index); diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index 0284be0e4e82..cdfe093e5c5c 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -235,7 +235,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, struct btrfs_device **device_out) { struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; - struct btrfs_device *device; + struct btrfs_device *device = NULL; + struct btrfs_device *tmp_device; struct file *bdev_file; struct block_device *bdev; u64 devid = BTRFS_DEV_REPLACE_DEVID; @@ -264,8 +265,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, sync_blockdev(bdev); - list_for_each_entry(device, &fs_devices->devices, dev_list) { - if (device->bdev == bdev) { + list_for_each_entry(tmp_device, &fs_devices->devices, dev_list) { + if (tmp_device->bdev == bdev) { btrfs_err(fs_info, "target device is in the filesystem!"); ret = -EEXIST; @@ -285,6 +286,7 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, device = btrfs_alloc_device(NULL, &devid, NULL, device_path); if (IS_ERR(device)) { ret = PTR_ERR(device); + device = NULL; goto error; } @@ -328,6 +330,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, error: /* Undo the open-time freeze deny. */ + if (device) + btrfs_free_device(device); btrfs_release_device_allow_freeze(bdev_file); return ret; } diff --git a/fs/btrfs/dir-item.c b/fs/btrfs/dir-item.c index 84f1c64423d3..3a90736915af 100644 --- a/fs/btrfs/dir-item.c +++ b/fs/btrfs/dir-item.c @@ -106,8 +106,11 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans, * Will return 0 or -ENOMEM */ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, - const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index) + const struct fscrypt_str *name, + struct btrfs_inode *dir, + const struct btrfs_key *location, u8 type, + u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; int ret2 = 0; @@ -119,17 +122,27 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, struct btrfs_key key; struct btrfs_disk_key disk_key; u32 data_size; + const bool need_delayed_index = (root != root->fs_info->tree_root); key.objectid = btrfs_ino(dir); key.type = BTRFS_DIR_ITEM_KEY; key.offset = btrfs_name_hash(name->name, name->len); path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; + if (!path) { + ret = -ENOMEM; + goto out_free_prealloc; + } btrfs_cpu_key_to_disk(&disk_key, location); + /* Pre-allocate the delayed dir index before modifying the btree. */ + if (need_delayed_index && !prealloc) { + prealloc = btrfs_prealloc_delayed_dir_index(dir, name->name, name->len); + if (IS_ERR(prealloc)) + return PTR_ERR(prealloc); + } + data_size = sizeof(*dir_item) + name->len; dir_item = insert_with_overflow(trans, root, path, &key, data_size, name->name, name->len); @@ -137,7 +150,7 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, ret = PTR_ERR(dir_item); if (ret == -EEXIST) goto second_insert; - goto out_free; + goto out_free_prealloc; } if (IS_ENCRYPTED(&dir->vfs_inode)) @@ -154,21 +167,21 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, write_extent_buffer(leaf, name->name, name_ptr, name->len); second_insert: - /* FIXME, use some real flag for selecting the extra index */ - if (root == root->fs_info->tree_root) { + if (!need_delayed_index) { ret = 0; - goto out_free; + goto out_free_prealloc; } btrfs_release_path(path); - ret2 = btrfs_insert_delayed_dir_index(trans, name->name, name->len, dir, - &disk_key, type, index); -out_free: + ret2 = btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, + &disk_key, type, index); if (ret) return ret; - if (ret2) - return ret2; - return 0; + return ret2; + +out_free_prealloc: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); + return ret; } static struct btrfs_dir_item *btrfs_lookup_match_dir( diff --git a/fs/btrfs/dir-item.h b/fs/btrfs/dir-item.h index e52174a8baf9..8d22976a0a77 100644 --- a/fs/btrfs/dir-item.h +++ b/fs/btrfs/dir-item.h @@ -13,12 +13,14 @@ struct btrfs_path; struct btrfs_inode; struct btrfs_root; struct btrfs_trans_handle; +struct btrfs_dir_index_prealloc; int btrfs_check_dir_item_collision(struct btrfs_root *root, u64 dir_ino, const struct fscrypt_str *name); int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index); + const struct btrfs_key *location, u8 type, u64 index, + struct btrfs_dir_index_prealloc *prealloc); struct btrfs_dir_item *btrfs_lookup_dir_item(struct btrfs_trans_handle *trans, struct btrfs_root *root, struct btrfs_path *path, u64 dir, @@ -53,5 +55,4 @@ static inline u64 btrfs_name_hash(const char *name, int len) { return crc32c((u32)~1, name, len); } - #endif diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index dc7ad92876c0..7fd8babd74a6 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -125,15 +125,13 @@ int btrfs_buffer_uptodate(struct extent_buffer *eb, u64 parent_transid, return 1; } - if (btrfs_header_generation(eb) != parent_transid) { - btrfs_err_rl(eb->fs_info, + btrfs_err_rl(eb->fs_info, "parent transid verify failed on logical %llu mirror %u wanted %llu found %llu", - eb->start, eb->read_mirror, - parent_transid, btrfs_header_generation(eb)); - clear_extent_buffer_uptodate(eb); - return 0; - } - return 1; + eb->start, eb->read_mirror, + parent_transid, btrfs_header_generation(eb)); + clear_extent_buffer_uptodate(eb); + + return 0; } static bool btrfs_supported_super_csum(u16 csum_type) @@ -176,19 +174,24 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, int mirror_num) { struct btrfs_fs_info *fs_info = eb->fs_info; - const u32 step = min(fs_info->nodesize, PAGE_SIZE); - const u32 nr_steps = eb->len / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; + struct btrfs_bio *bbio; + int ret; if (sb_rdonly(fs_info->sb)) return -EROFS; + /* + * This bbio is only to queue all pages for btrfs_repair_bbio_failure(). + * Thus it will never get its endio called. + */ + bbio = btrfs_bio_alloc(max(1, fs_info->nodesize >> PAGE_SHIFT), REQ_OP_READ, + BTRFS_I(fs_info->btree_inode), eb->start, NULL, NULL); + bbio->bio.bi_iter.bi_sector = eb->start >> SECTOR_SHIFT; for (int i = 0; i < num_extent_pages(eb); i++) { struct folio *folio = eb->folios[i]; /* No large folio support yet. */ ASSERT(folio_order(folio) == 0); - ASSERT(i < nr_steps); /* * For nodesize < page size, there is just one paddr, with some @@ -197,11 +200,17 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, * For nodesize >= page size, it's one or more paddrs, and eb->start * must be aligned to page boundary. */ - paddrs[i] = page_to_phys(&folio->page) + offset_in_page(eb->start); + ret = bio_add_page(&bbio->bio, &folio->page, min(PAGE_SIZE, fs_info->nodesize), + offset_in_page(eb->start)); + ASSERT(ret == min(PAGE_SIZE, fs_info->nodesize)); } + /* Since the bbio is never submitted, we have to save the iter manually. */ + bbio->saved_iter = bbio->bio.bi_iter; - return btrfs_repair_io_failure(fs_info, 0, eb->start, eb->len, - eb->start, paddrs, step, mirror_num); + ret = btrfs_repair_bbio_failure(bbio, &bbio->saved_iter, fs_info->nodesize, + mirror_num); + bio_put(&bbio->bio); + return ret; } /* @@ -1485,7 +1494,9 @@ static int cleaner_kthread(void *arg) btrfs_run_delayed_iputs(fs_info); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); again = btrfs_clean_one_deleted_snapshot(fs_info); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); mutex_unlock(&fs_info->cleaner_mutex); /* @@ -1769,7 +1780,6 @@ static void btrfs_stop_all_workers(struct btrfs_fs_info *fs_info) if (fs_info->rmw_workers) destroy_workqueue(fs_info->rmw_workers); btrfs_destroy_workqueue(fs_info->endio_write_workers); - btrfs_destroy_workqueue(fs_info->endio_freespace_worker); btrfs_destroy_workqueue(fs_info->delayed_workers); btrfs_destroy_workqueue(fs_info->caching_workers); btrfs_destroy_workqueue(fs_info->flush_workers); @@ -1980,9 +1990,6 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) fs_info->endio_write_workers = btrfs_alloc_workqueue(fs_info, "endio-write", flags, max_active, 2); - fs_info->endio_freespace_worker = - btrfs_alloc_workqueue(fs_info, "freespace-write", flags, - max_active, 0); fs_info->delayed_workers = btrfs_alloc_workqueue(fs_info, "delayed-meta", flags, max_active, 0); @@ -1995,8 +2002,7 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) if (!(fs_info->workers && fs_info->delalloc_workers && fs_info->flush_workers && fs_info->endio_workers && fs_info->endio_meta_workers && - fs_info->endio_write_workers && - fs_info->endio_freespace_worker && fs_info->rmw_workers && + fs_info->endio_write_workers && fs_info->rmw_workers && fs_info->caching_workers && fs_info->fixup_workers && fs_info->delayed_workers && fs_info->qgroup_rescan_workers && fs_info->discard_ctl.discard_workers)) { @@ -2374,7 +2380,7 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info, } ret = btrfs_check_chunk_valid(fs_info, NULL, chunk, key.offset, sectorsize); - if (ret < 0) + if (unlikely(ret < 0)) return ret; cur += btrfs_chunk_item_size(num_stripes); } @@ -3063,7 +3069,6 @@ static int btrfs_cleanup_fs_roots(struct btrfs_fs_info *fs_info) int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) { int ret; - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); bool rebuild_free_space_tree = false; if (btrfs_test_opt(fs_info, CLEAR_CACHE) && @@ -3158,8 +3163,8 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) } } - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) { - ret = btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); + if (btrfs_free_space_cache_v1_active(fs_info)) { + ret = btrfs_cleanup_free_space_cache_v1(fs_info); if (ret) return ret; } @@ -3275,20 +3280,6 @@ int btrfs_check_features(struct btrfs_fs_info *fs_info, bool is_rw_mount) return -EINVAL; } - /* - * Subpage/bs > ps runtime limitation on v1 cache. - * - * V1 space cache still has some hard coded PAGE_SIZE usage, while - * we're already defaulting to v2 cache, no need to bother v1 as it's - * going to be deprecated anyway. - */ - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_warn(fs_info, - "v1 space cache is not supported for page size %lu with sectorsize %u", - PAGE_SIZE, fs_info->sectorsize); - return -EINVAL; - } - /* This can be called by remount, we need to protect the super block. */ spin_lock(&fs_info->super_lock); btrfs_set_super_incompat_flags(disk_super, incompat); @@ -4217,8 +4208,6 @@ int write_all_supers(struct btrfs_trans_handle *trans) total_errors++; } if (unlikely(total_errors > max_errors)) { - btrfs_err(fs_info, "%d errors while writing supers", - total_errors); mutex_unlock(&fs_info->fs_devices->device_list_mutex); /* FUA is masked off if unsupported and can't be the reason */ @@ -4446,9 +4435,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * to finish an ordered extent - end_bbio_compressed_write() * calls btrfs_finish_ordered_extent() which in turns does a call to * btrfs_queue_ordered_fn(), and that queues the ordered extent - * completion either in the endio_write_workers work queue or in the - * fs_info->endio_freespace_worker work queue. We flush those queues - * below, so before we flush them we must flush this queue for the + * completion in the endio_write_workers work queue. We flush that + * queue below, so before we flush it we must flush this queue for the * workers of compressed writes. */ flush_workqueue(fs_info->endio_workers); @@ -4474,8 +4462,6 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * btrfs_finish_ordered_io() when we are unmounting). */ btrfs_flush_workqueue(fs_info->endio_write_workers); - /* Ordered extents for free space inodes. */ - btrfs_flush_workqueue(fs_info->endio_freespace_worker); /* * Run delayed iputs in case an async reclaim worker is waiting for them * to be run as mentioned above. @@ -4864,26 +4850,6 @@ static void btrfs_destroy_pinned_extent(struct btrfs_fs_info *fs_info, } } -static void btrfs_cleanup_bg_io(struct btrfs_block_group *cache) -{ - struct inode *inode; - - inode = cache->io_ctl.inode; - if (inode) { - unsigned int nofs_flag; - - nofs_flag = memalloc_nofs_save(); - invalidate_inode_pages2(inode->i_mapping); - memalloc_nofs_restore(nofs_flag); - - BTRFS_I(inode)->generation = 0; - cache->io_ctl.inode = NULL; - iput(inode); - } - ASSERT(cache->io_ctl.pages == NULL); - btrfs_put_block_group(cache); -} - void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_fs_info *fs_info) { @@ -4895,40 +4861,13 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_block_group, dirty_list); - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_cleanup_bg_io(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - list_del_init(&cache->dirty_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - spin_unlock(&cur_trans->dirty_bgs_lock); btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(&cur_trans->io_bgs)) { - cache = list_first_entry(&cur_trans->io_bgs, - struct btrfs_block_group, - io_list); - - list_del_init(&cache->io_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - btrfs_cleanup_bg_io(cache); - } } static void btrfs_free_all_qgroup_pertrans(struct btrfs_fs_info *fs_info) @@ -4964,7 +4903,6 @@ void btrfs_cleanup_one_transaction(struct btrfs_transaction *cur_trans) btrfs_cleanup_dirty_bgs(cur_trans, fs_info); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); list_for_each_entry_safe(dev, tmp, &cur_trans->dev_update_list, post_commit_list) { diff --git a/fs/btrfs/extent-io-tree.c b/fs/btrfs/extent-io-tree.c index d6df11f6088c..992b8b42bdb4 100644 --- a/fs/btrfs/extent-io-tree.c +++ b/fs/btrfs/extent-io-tree.c @@ -751,7 +751,7 @@ hit_next: btrfs_split_delalloc_extent(tree->inode, state, start); /* - * Temporarilly ajdust this state's range to match the + * Temporarily ajdust this state's range to match the * range for which we are clearing bits. */ state->start = start; diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c index d6a4390ee34a..a0d5ab03aae2 100644 --- a/fs/btrfs/extent-tree.c +++ b/fs/btrfs/extent-tree.c @@ -6315,6 +6315,16 @@ int btrfs_drop_snapshot(struct btrfs_root *root, bool update_ref, bool for_reloc set_bit(BTRFS_ROOT_DELETING, &root->state); unfinished_drop = test_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state); + /* + * For subvolume dropping, check if the subvolume is large enough so + * that we need to mark qgroup inconsistent to avoid long qgroup stall. + * + * Even for a subvolume without any snapshot, there can still be + * a lot of qgroup records queued into one transaction. + */ + if (!for_reloc) + btrfs_qgroup_check_tree_drop(fs_info, rootid, + btrfs_header_level(root->node)); if (btrfs_disk_key_objectid(&root_item->drop_progress) == 0) { level = btrfs_header_level(root->node); path->nodes[level] = btrfs_lock_root_node(root); diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index d7600e5fa3d9..55e9144d4759 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -110,14 +110,14 @@ struct btrfs_bio_ctrl { * make the decision when submitting the bio. * * The pattern between do_readpage(), submit_one_bio() and - * submit_extent_folio() is quite subtle, so tracking this is tricky. + * submit_one_block() is quite subtle, so tracking this is tricky. * * As we process extent E, we might submit a bio with existing built up * extents before adding E to a new bio, or we might just add E to the * bio. As a result, E's generation could apply to the current bio or * to the next one, so we need to be careful to update the bio_ctrl's * generation with E's only when we are sure E is added to bio_ctrl->bbio - * in submit_extent_folio(). + * in submit_one_block(). * * See the comment in btrfs_lookup_bio_sums() for more detail on the * need for this optimization. @@ -797,118 +797,95 @@ static int alloc_new_bio(struct btrfs_inode *inode, } /* - * @disk_bytenr: logical bytenr where the write will be - * @page: page to add to the bio - * @size: portion of page that we want to write to - * @pg_offset: offset of the new bio or to check whether we are adding - * a contiguous page to the previous one + * @disk_bytenr: logical bytenr where the read/write will be + * @folio: the folio the block belongs to + * @pg_offset: the offset inside the folio * @read_em_generation: generation of the extent_map we are submitting * (only used for read) * - * The will either add the page into the existing @bio_ctrl->bbio, or allocate a + * This will either add the block into the existing @bio_ctrl->bbio, or allocate a * new one in @bio_ctrl->bbio. * The mirror number for this IO should already be initialized in * @bio_ctrl->mirror_num. * - * Return the number of bytes that are queued into a bio. - * If the returned bytes is smaller than @size, it means we hit a critical error - * for data write, where there is no ordered extent for the range. + * Return 0 if the block is queued or submitted. + * Return <0 for error. */ -static unsigned int submit_extent_folio(struct btrfs_bio_ctrl *bio_ctrl, - u64 disk_bytenr, struct folio *folio, - size_t size, unsigned long pg_offset, - u64 read_em_generation) +static int submit_one_block(struct btrfs_bio_ctrl *bio_ctrl, + u64 disk_bytenr, struct folio *folio, + unsigned long pg_offset, u64 read_em_generation) { struct btrfs_inode *inode = folio_to_inode(folio); + const struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; loff_t file_offset = folio_pos(folio) + pg_offset; - unsigned int queued = 0; - ASSERT(pg_offset + size <= folio_size(folio)); + ASSERT(pg_offset + blocksize <= folio_size(folio)); ASSERT(bio_ctrl->end_io_func); if (bio_ctrl->bbio && !btrfs_bio_is_contig(bio_ctrl, disk_bytenr, file_offset)) submit_one_bio(bio_ctrl); - do { - u32 len = size; - - /* Allocate new bio if needed */ - if (!bio_ctrl->bbio) { - int ret; - - ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); - if (ret < 0) - break; - } - - /* Cap to the current ordered extent boundary if there is one. */ - if (len > bio_ctrl->len_to_oe_boundary) { - ASSERT(bio_ctrl->compress_type == BTRFS_COMPRESS_NONE); - ASSERT(is_data_inode(inode)); - len = bio_ctrl->len_to_oe_boundary; - } +again: + /* Allocate new bio if needed */ + if (!bio_ctrl->bbio) { + int ret; - if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, len, pg_offset)) { - /* bio full: move on to a new one */ - submit_one_bio(bio_ctrl); - continue; - } - /* - * Now that the folio is definitely added to the bio, include its - * generation in the max generation calculation. - */ - bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); - bio_ctrl->next_file_offset += len; + ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); + if (ret < 0) + return ret; + } - if (bio_ctrl->wbc) - wbc_account_cgroup_owner(bio_ctrl->wbc, folio, len); + if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, blocksize, pg_offset)) { + /* bio full: move on to a new one */ + submit_one_bio(bio_ctrl); + goto again; + } - size -= len; - pg_offset += len; - disk_bytenr += len; - file_offset += len; - queued += len; + /* + * Now that the folio is definitely added to the bio, include its + * generation in the max generation calculation. + */ + bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); + bio_ctrl->next_file_offset += blocksize; - /* - * len_to_oe_boundary defaults to U32_MAX, which isn't folio or - * sector aligned. alloc_new_bio() then sets it to the end of - * our ordered extent for writes into zoned devices. - * - * When len_to_oe_boundary is tracking an ordered extent, we - * trust the ordered extent code to align things properly, and - * the check above to cap our write to the ordered extent - * boundary is correct. - * - * When len_to_oe_boundary is U32_MAX, the cap above would - * result in a 4095 byte IO for the last folio right before - * we hit the bio limit of UINT_MAX. bio_add_folio() has all - * the checks required to make sure we don't overflow the bio, - * and we should just ignore len_to_oe_boundary completely - * unless we're using it to track an ordered extent. - * - * It's pretty hard to make a bio sized U32_MAX, but it can - * happen when the page cache is able to feed us contiguous - * folios for large extents. - */ - if (bio_ctrl->len_to_oe_boundary != U32_MAX) - bio_ctrl->len_to_oe_boundary -= len; + if (bio_ctrl->wbc) + wbc_account_cgroup_owner(bio_ctrl->wbc, folio, blocksize); - /* Ordered extent boundary: move on to a new bio. */ - if (bio_ctrl->len_to_oe_boundary == 0) - submit_one_bio(bio_ctrl); - /* - * If we have accumulated decent amount of IO, send it to the - * block layer so that IO can run while we are accumulating - * more folios to write. - */ - else if (bio_ctrl->wbc && - bio_ctrl->bbio->bio.bi_iter.bi_size >= - inode->root->fs_info->writeback_bio_size) - submit_one_bio(bio_ctrl); + /* + * len_to_oe_boundary defaults to U32_MAX, which isn't folio or sector + * aligned. alloc_new_bio() then sets it to the end of our ordered + * extent for writes into zoned devices. + * + * When len_to_oe_boundary is tracking an ordered extent, the + * len_to_oe_boundary should follow that OE and never go beyond the max + * extent size (128MiB). + * + * When len_to_oe_boundary is U32_MAX, decreasing the length by + * blocksize will never make it reach 0, thus skipping the later + * submit_one_bio() call. So if len_to_oe_boundary() is not tracking + * an OE, do not decrease it. + * + * It's pretty hard to make a bio sized U32_MAX, but it can happen when + * the page cache is able to feed us contiguous folios for large + * extents. + */ + if (bio_ctrl->len_to_oe_boundary != U32_MAX) + bio_ctrl->len_to_oe_boundary -= blocksize; - } while (size); - return queued; + /* Ordered extent boundary: move on to a new bio. */ + if (bio_ctrl->len_to_oe_boundary == 0) + submit_one_bio(bio_ctrl); + /* + * If we have accumulated decent amount of IO, send it to the block + * layer so that IO can run while we are accumulating more folios to + * write. + */ + else if (bio_ctrl->wbc && + bio_ctrl->bbio->bio.bi_iter.bi_size >= fs_info->writeback_bio_size) + submit_one_bio(bio_ctrl); + return 0; } static int attach_extent_buffer_folio(struct extent_buffer *eb, @@ -1092,7 +1069,6 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, u64 disk_bytenr; u64 block_start; u64 em_gen; - unsigned int queued; ASSERT(IS_ALIGNED(cur, fs_info->sectorsize)); if (cur >= last_byte) { @@ -1206,10 +1182,9 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, if (force_bio_submit) submit_one_bio(bio_ctrl); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, blocksize, - pg_offset, em_gen); + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, pg_offset, em_gen); /* Read submission should not fail. */ - ASSERT(queued == blocksize); + ASSERT(ret == 0); } return 0; } @@ -1808,33 +1783,51 @@ out: return 0; } +static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbio, + u64 filepos) +{ + struct btrfs_ordered_extent *oe; + + if (!bbio || !bbio->ordered) + return NULL; + + oe = bbio->ordered; + if (!in_range(filepos, oe->file_offset, oe->num_bytes)) + return NULL; + + refcount_inc(&oe->refs); + return oe; +} + /* * Return 0 if we have submitted or queued the sector for submission. * Return <0 for critical errors, and the involved sector will be cleaned up. * * Caller should make sure filepos < i_size and handle filepos >= i_size case. */ -static int submit_one_sector(struct btrfs_inode *inode, - struct folio *folio, - u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, - loff_t i_size) +static int submit_write_sector(struct btrfs_inode *inode, + struct folio *folio, + u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, + loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct extent_map *em; + struct btrfs_ordered_extent *oe; u64 block_start; u64 disk_bytenr; u64 extent_offset; - u64 em_end; const u32 sectorsize = fs_info->sectorsize; - unsigned int queued; + int ret; ASSERT(IS_ALIGNED(filepos, sectorsize)); /* @filepos >= i_size case should be handled by the caller. */ ASSERT(filepos < i_size); - em = btrfs_get_extent(inode, NULL, filepos, sectorsize); - if (IS_ERR(em)) { + /* Try to reuse the existing OE from bbio first. */ + oe = get_oe_from_bbio(bio_ctrl->bbio, filepos); + if (!oe) + oe = btrfs_lookup_ordered_extent(inode, filepos); + if (unlikely(!oe)) { /* * bio_ctrl may contain a bio crossing several folios. * Submit it immediately so that the bio has a chance @@ -1857,31 +1850,25 @@ static int submit_one_sector(struct btrfs_inode *inode, */ btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); - return PTR_ERR(em); + btrfs_err_rl(fs_info, + "no ordered extent for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), btrfs_ino(inode), + filepos); + return -EUCLEAN; } - extent_offset = filepos - em->start; - em_end = btrfs_extent_map_end(em); - ASSERT(filepos <= em_end); - ASSERT(IS_ALIGNED(em->start, sectorsize)); - ASSERT(IS_ALIGNED(em->len, sectorsize)); + extent_offset = filepos - oe->file_offset; + ASSERT(filepos < oe->file_offset + oe->num_bytes); + ASSERT(IS_ALIGNED(oe->file_offset, sectorsize)); + ASSERT(IS_ALIGNED(oe->num_bytes, sectorsize)); + ASSERT(oe->compress_type == BTRFS_COMPRESS_NONE); + ASSERT(!test_bit(BTRFS_ORDERED_COMPRESSED, &oe->flags)); - block_start = btrfs_extent_map_block_start(em); - disk_bytenr = btrfs_extent_map_block_start(em) + extent_offset; + block_start = oe->disk_bytenr + oe->offset; + disk_bytenr = block_start + extent_offset; - ASSERT(!btrfs_extent_map_is_compressed(em)); - ASSERT(block_start != EXTENT_MAP_HOLE); - ASSERT(block_start != EXTENT_MAP_INLINE); + btrfs_put_ordered_extent(oe); - btrfs_free_extent_map(em); - em = NULL; - - /* - * Although the PageDirty bit is cleared before entering this - * function, subpage dirty bit is not cleared. - * So clear subpage dirty bit here so next time we won't submit - * a folio for a range already written to disk. - */ btrfs_folio_clear_dirty(fs_info, folio, filepos, sectorsize); btrfs_folio_set_writeback(fs_info, folio, filepos, sectorsize); /* @@ -1892,13 +1879,17 @@ static int submit_one_sector(struct btrfs_inode *inode, */ ASSERT(folio_test_writeback(folio)); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, - sectorsize, filepos - folio_pos(folio), 0); - if (unlikely(queued < sectorsize)) { + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, + offset_in_folio(folio, filepos), 0); + if (unlikely(ret < 0)) { btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); - return -EUCLEAN; + btrfs_err_rl(fs_info, + "failed to queue sector for root %lld ino %llu filepos %llu: %pe", + btrfs_root_id(inode->root), + btrfs_ino(inode), filepos, ERR_PTR(ret)); + return ret; } return 0; } @@ -1983,7 +1974,7 @@ static noinline_for_stack int extent_writepage_io(struct btrfs_inode *inode, btrfs_folio_clear_dirty(fs_info, folio, cur, fs_info->sectorsize); continue; } - ret = submit_one_sector(inode, folio, cur, bio_ctrl, i_size); + ret = submit_write_sector(inode, folio, cur, bio_ctrl, i_size); if (unlikely(ret < 0)) { if (!found_error) found_error = ret; diff --git a/fs/btrfs/extent_map.c b/fs/btrfs/extent_map.c index 6ad7b39ae358..86d9c6f5ff4b 100644 --- a/fs/btrfs/extent_map.c +++ b/fs/btrfs/extent_map.c @@ -1220,6 +1220,14 @@ static struct btrfs_inode *find_first_inode_to_shrink(struct btrfs_root *root, tree = &inode->extent_tree; /* + * Most inodes have no extent maps, so check without the lock. + * The race is harmless: a false empty just defers the inode to + * a later scan, and a false non-empty is caught under the lock. + */ + if (data_race(RB_EMPTY_ROOT(&tree->root))) + goto next; + + /* * We want to be fast so if the lock is busy we don't want to * spend time waiting for it (some task is about to do IO for * the inode). diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index cf50fd623f41..ff8f8cad00fc 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -397,17 +397,6 @@ int btrfs_lookup_bio_sums(struct btrfs_bio *bbio) path->reada = READA_FORWARD; /* - * the free space stuff is only read when it hasn't been - * updated in the current transaction. So, we can safely - * read from the commit root and sidestep a nasty deadlock - * between reading the free space cache and updating the csum tree. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - - /* * If we are searching for a csum of an extent from a past * transaction, we can search in the commit root and reduce * lock contention on the csum tree extent buffers. @@ -797,29 +786,20 @@ fail: return ret; } -static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) +static void csum_one_bio(struct btrfs_bio *bbio) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct bio *bio = &bbio->bio; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter = *src; - phys_addr_t paddr; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - u32 offset = 0; int index = 0; - btrfs_bio_for_each_block(paddr, bio, &iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; + for (struct bvec_iter *iter = &bbio->csum_saved_iter; + iter->bi_size; + bio_advance_iter(&bbio->bio, iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, iter, sums->sums + index); - if (IS_ALIGNED(offset, blocksize)) { - btrfs_calculate_block_csum_pages(fs_info, paddrs, sums->sums + index); - index += fs_info->csum_size; - } + index += fs_info->csum_size; } } @@ -828,9 +808,8 @@ static void csum_one_bio_work(struct work_struct *work) struct btrfs_bio *bbio = container_of(work, struct btrfs_bio, csum_work); ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE); - ASSERT(bbio->async_csum == true); - csum_one_bio(bbio, &bbio->csum_saved_iter); - complete(&bbio->csum_done); + csum_one_bio(bbio); + bio_endio(&bbio->bio); } /* @@ -859,15 +838,14 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async) bbio->sums = sums; btrfs_add_ordered_sum(ordered, sums); + bbio->csum_saved_iter = bio->bi_iter; if (!async) { - csum_one_bio(bbio, &bbio->bio.bi_iter); + csum_one_bio(bbio); return 0; } - init_completion(&bbio->csum_done); - bbio->async_csum = true; - bbio->csum_saved_iter = bbio->bio.bi_iter; + bio_inc_remaining(bio); INIT_WORK(&bbio->csum_work, csum_one_bio_work); - schedule_work(&bbio->csum_work); + queue_work(fs_info->endio_workers, &bbio->csum_work); return 0; } diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index e2af75a205ea..2a40167c3fb6 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -9,7 +9,6 @@ #include <linux/slab.h> #include <linux/math64.h> #include <linux/ratelimit.h> -#include <linux/error-injection.h> #include <linux/sched/mm.h> #include <linux/string_choices.h> #include "extent-tree.h" @@ -23,7 +22,6 @@ #include "space-info.h" #include "block-group.h" #include "discard.h" -#include "subpage.h" #include "inode-item.h" #include "accessors.h" #include "file-item.h" @@ -38,12 +36,6 @@ static struct kmem_cache *btrfs_free_space_cachep; static struct kmem_cache *btrfs_free_space_bitmap_cachep; -struct btrfs_trim_range { - u64 start; - u64 bytes; - struct list_head list; -}; - static int link_free_space(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info); static void unlink_free_space(struct btrfs_free_space_ctl *ctl, @@ -57,11 +49,6 @@ static void bitmap_clear_bits(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info, u64 offset, u64 bytes, bool update_stats); -static void btrfs_crc32c_final(u32 crc, u8 *result) -{ - put_unaligned_le32(~crc, result); -} - static void __btrfs_remove_free_space_cache(struct btrfs_free_space_ctl *ctl) { struct btrfs_free_space *info; @@ -123,10 +110,6 @@ static struct inode *__lookup_free_space_inode(struct btrfs_root *root, if (IS_ERR(inode)) return ERR_CAST(inode); - mapping_set_gfp_mask(inode->vfs_inode.i_mapping, - mapping_gfp_constraint(inode->vfs_inode.i_mapping, - ~(__GFP_FS | __GFP_HIGHMEM))); - return &inode->vfs_inode; } @@ -135,7 +118,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, { struct btrfs_fs_info *fs_info = block_group->fs_info; struct inode *inode = NULL; - u32 flags = BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; spin_lock(&block_group->lock); if (block_group->inode) @@ -150,13 +132,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; spin_lock(&block_group->lock); - if (!((BTRFS_I(inode)->flags & flags) == flags)) { - btrfs_info(fs_info, "Old style space inode found, converting."); - BTRFS_I(inode)->flags |= BTRFS_INODE_NODATASUM | - BTRFS_INODE_NODATACOW; - block_group->disk_cache_state = BTRFS_DC_CLEAR; - } - if (!test_and_set_bit(BLOCK_GROUP_FLAG_IREF, &block_group->runtime_flags)) block_group->inode = BTRFS_I(igrab(inode)); spin_unlock(&block_group->lock); @@ -164,78 +139,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; } -static int __create_free_space_inode(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_path *path, - u64 ino, u64 offset) -{ - struct btrfs_key key; - struct btrfs_disk_key disk_key; - struct btrfs_free_space_header *header; - struct btrfs_inode_item *inode_item; - struct extent_buffer *leaf; - /* We inline CRCs for the free disk space cache */ - const u64 flags = BTRFS_INODE_NOCOMPRESS | BTRFS_INODE_PREALLOC | - BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; - int ret; - - ret = btrfs_insert_empty_inode(trans, root, path, ino); - if (ret) - return ret; - - leaf = path->nodes[0]; - inode_item = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_inode_item); - btrfs_item_key(leaf, &disk_key, path->slots[0]); - memzero_extent_buffer(leaf, (unsigned long)inode_item, - sizeof(*inode_item)); - btrfs_set_inode_generation(leaf, inode_item, trans->transid); - btrfs_set_inode_size(leaf, inode_item, 0); - btrfs_set_inode_nbytes(leaf, inode_item, 0); - btrfs_set_inode_uid(leaf, inode_item, 0); - btrfs_set_inode_gid(leaf, inode_item, 0); - btrfs_set_inode_mode(leaf, inode_item, S_IFREG | 0600); - btrfs_set_inode_flags(leaf, inode_item, flags); - btrfs_set_inode_nlink(leaf, inode_item, 1); - btrfs_set_inode_transid(leaf, inode_item, trans->transid); - btrfs_set_inode_block_group(leaf, inode_item, offset); - btrfs_release_path(path); - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - ret = btrfs_insert_empty_item(trans, root, path, &key, - sizeof(struct btrfs_free_space_header)); - if (ret < 0) { - btrfs_release_path(path); - return ret; - } - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - memzero_extent_buffer(leaf, (unsigned long)header, sizeof(*header)); - btrfs_set_free_space_key(leaf, header, &disk_key); - btrfs_release_path(path); - - return 0; -} - -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - int ret; - u64 ino; - - ret = btrfs_get_free_objectid(trans->fs_info->tree_root, &ino); - if (ret < 0) - return ret; - - return __create_free_space_inode(trans->fs_info->tree_root, trans, path, - ino, block_group->start); -} - /* * inode is an optional sink: if it is NULL, btrfs_remove_free_space_inode * handles lookup, otherwise it takes ownership and iputs the inode. @@ -292,7 +195,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, } int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *vfs_inode) { struct btrfs_truncate_control control = { @@ -306,33 +208,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, struct btrfs_root *root = inode->root; struct extent_state *cached_state = NULL; int ret = 0; - bool locked = false; - - if (block_group) { - BTRFS_PATH_AUTO_FREE(path); - - path = btrfs_alloc_path(); - if (!path) { - ret = -ENOMEM; - goto fail; - } - locked = true; - mutex_lock(&trans->transaction->cache_write_mutex); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - } - - /* - * now that we've truncated the cache away, its no longer - * setup or written - */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - } btrfs_i_size_write(inode, 0); truncate_pagecache(vfs_inode, 0); @@ -356,336 +231,12 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, ret = btrfs_update_inode(trans, inode); fail: - if (locked) - mutex_unlock(&trans->transaction->cache_write_mutex); if (ret) btrfs_abort_transaction(trans, ret); return ret; } -static void readahead_cache(struct inode *inode) -{ - struct file_ra_state ra; - pgoff_t last_index; - - file_ra_state_init(&ra, inode->i_mapping); - last_index = (i_size_read(inode) - 1) >> PAGE_SHIFT; - - page_cache_sync_readahead(inode->i_mapping, &ra, NULL, 0, last_index); -} - -static int io_ctl_init(struct btrfs_io_ctl *io_ctl, struct inode *inode, - int write) -{ - int num_pages; - - num_pages = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); - - /* Make sure we can fit our crcs and generation into the first page */ - if (write && (num_pages * sizeof(u32) + sizeof(u64)) > PAGE_SIZE) - return -ENOSPC; - - memset(io_ctl, 0, sizeof(struct btrfs_io_ctl)); - - io_ctl->pages = kzalloc_objs(struct page *, num_pages, GFP_NOFS); - if (!io_ctl->pages) - return -ENOMEM; - - io_ctl->num_pages = num_pages; - io_ctl->fs_info = inode_to_fs_info(inode); - io_ctl->inode = inode; - - return 0; -} -ALLOW_ERROR_INJECTION(io_ctl_init, ERRNO); - -static void io_ctl_free(struct btrfs_io_ctl *io_ctl) -{ - kfree(io_ctl->pages); - io_ctl->pages = NULL; -} - -static void io_ctl_unmap_page(struct btrfs_io_ctl *io_ctl) -{ - if (io_ctl->cur) { - io_ctl->cur = NULL; - io_ctl->orig = NULL; - } -} - -static void io_ctl_map_page(struct btrfs_io_ctl *io_ctl, int clear) -{ - ASSERT(io_ctl->index < io_ctl->num_pages); - io_ctl->page = io_ctl->pages[io_ctl->index++]; - io_ctl->cur = page_address(io_ctl->page); - io_ctl->orig = io_ctl->cur; - io_ctl->size = PAGE_SIZE; - if (clear) - clear_page(io_ctl->cur); -} - -static void io_ctl_drop_pages(struct btrfs_io_ctl *io_ctl) -{ - int i; - - io_ctl_unmap_page(io_ctl); - - for (i = 0; i < io_ctl->num_pages; i++) { - if (io_ctl->pages[i]) { - unlock_page(io_ctl->pages[i]); - put_page(io_ctl->pages[i]); - } - } -} - -static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate) -{ - struct folio *folio; - struct inode *inode = io_ctl->inode; - gfp_t mask = btrfs_alloc_write_mask(inode->i_mapping); - int i; - - for (i = 0; i < io_ctl->num_pages; i++) { - int ret; - - folio = __filemap_get_folio(inode->i_mapping, i, - FGP_LOCK | FGP_ACCESSED | FGP_CREAT, - mask); - if (IS_ERR(folio)) { - io_ctl_drop_pages(io_ctl); - return PTR_ERR(folio); - } - - ret = set_folio_extent_mapped(folio); - if (ret < 0) { - folio_unlock(folio); - folio_put(folio); - io_ctl_drop_pages(io_ctl); - return ret; - } - - io_ctl->pages[i] = &folio->page; - if (uptodate && !folio_test_uptodate(folio)) { - btrfs_read_folio(NULL, folio); - folio_lock(folio); - if (folio->mapping != inode->i_mapping) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "free space cache page truncated"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - if (!folio_test_uptodate(folio)) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "error reading free space cache"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - } - } - - for (i = 0; i < io_ctl->num_pages; i++) - clear_page_dirty_for_io(io_ctl->pages[i]); - - return 0; -} - -static void io_ctl_set_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - io_ctl_map_page(io_ctl, 1); - - /* - * Skip the csum areas. If we don't check crcs then we just have a - * 64bit chunk at the front of the first page. - */ - io_ctl->cur += (sizeof(u32) * io_ctl->num_pages); - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - put_unaligned_le64(generation, io_ctl->cur); - io_ctl->cur += sizeof(u64); -} - -static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - u64 cache_gen; - - /* - * Skip the crc area. If we don't check crcs then we just have a 64bit - * chunk at the front of the first page. - */ - io_ctl->cur += sizeof(u32) * io_ctl->num_pages; - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - cache_gen = get_unaligned_le64(io_ctl->cur); - if (cache_gen != generation) { - btrfs_err_rl(io_ctl->fs_info, - "space cache generation (%llu) does not match inode (%llu)", - cache_gen, generation); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - io_ctl->cur += sizeof(u64); - return 0; -} - -static void io_ctl_set_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - io_ctl_unmap_page(io_ctl); - tmp = page_address(io_ctl->pages[0]); - tmp += index; - *tmp = crc; -} - -static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp, val; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index >= io_ctl->num_pages) - return -EIO; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - tmp = page_address(io_ctl->pages[0]); - tmp += index; - val = *tmp; - - io_ctl_map_page(io_ctl, 0); - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - if (val != crc) { - btrfs_err_rl(io_ctl->fs_info, - "csum mismatch on free space cache"); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - - return 0; -} - -static int io_ctl_add_entry(struct btrfs_io_ctl *io_ctl, u64 offset, u64 bytes, - void *bitmap) -{ - struct btrfs_free_space_entry *entry; - - if (!io_ctl->cur) - return -ENOSPC; - - entry = io_ctl->cur; - put_unaligned_le64(offset, &entry->offset); - put_unaligned_le64(bytes, &entry->bytes); - entry->type = (bitmap) ? BTRFS_FREE_SPACE_BITMAP : - BTRFS_FREE_SPACE_EXTENT; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - - /* No more pages to map */ - if (io_ctl->index >= io_ctl->num_pages) - return 0; - - /* map the next page */ - io_ctl_map_page(io_ctl, 1); - return 0; -} - -static int io_ctl_add_bitmap(struct btrfs_io_ctl *io_ctl, void *bitmap) -{ - if (!io_ctl->cur) - return -ENOSPC; - - /* - * If we aren't at the start of the current page, unmap this one and - * map the next one if there is any left. - */ - if (io_ctl->cur != io_ctl->orig) { - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index >= io_ctl->num_pages) - return -ENOSPC; - io_ctl_map_page(io_ctl, 0); - } - - copy_page(io_ctl->cur, bitmap); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index < io_ctl->num_pages) - io_ctl_map_page(io_ctl, 0); - return 0; -} - -static void io_ctl_zero_remaining_pages(struct btrfs_io_ctl *io_ctl) -{ - /* - * If we're not on the boundary we know we've modified the page and we - * need to crc the page. - */ - if (io_ctl->cur != io_ctl->orig) - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - else - io_ctl_unmap_page(io_ctl); - - while (io_ctl->index < io_ctl->num_pages) { - io_ctl_map_page(io_ctl, 1); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - } -} - -static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry, u8 *type) -{ - struct btrfs_free_space_entry *e; - int ret; - - if (!io_ctl->cur) { - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - } - - e = io_ctl->cur; - entry->offset = get_unaligned_le64(&e->offset); - entry->bytes = get_unaligned_le64(&e->bytes); - *type = e->type; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_unmap_page(io_ctl); - - return 0; -} - -static int io_ctl_read_bitmap(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry) -{ - int ret; - - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - - copy_page(entry->bitmap, io_ctl->cur); - io_ctl_unmap_page(io_ctl); - - return 0; -} - static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) { struct btrfs_block_group *block_group = ctl->block_group; @@ -731,824 +282,6 @@ static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) div_u64(extent_bytes, sizeof(struct btrfs_free_space)); } -static int __load_free_space_cache(struct btrfs_root *root, struct inode *inode, - struct btrfs_free_space_ctl *ctl, - struct btrfs_path *path, u64 offset) -{ - struct btrfs_fs_info *fs_info = root->fs_info; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - struct btrfs_io_ctl io_ctl; - struct btrfs_key key; - struct btrfs_free_space *e, *n; - LIST_HEAD(bitmaps); - u64 num_entries; - u64 num_bitmaps; - u64 generation; - u8 type; - int ret = 0; - - /* Nothing in the space cache, goodbye */ - if (!i_size_read(inode)) - return 0; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(NULL, root, &key, path, 0, 0); - if (ret < 0) - return 0; - else if (ret > 0) { - btrfs_release_path(path); - return 0; - } - - ret = -1; - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - num_entries = btrfs_free_space_entries(leaf, header); - num_bitmaps = btrfs_free_space_bitmaps(leaf, header); - generation = btrfs_free_space_generation(leaf, header); - btrfs_release_path(path); - - if (!BTRFS_I(inode)->generation) { - btrfs_info(fs_info, - "the free space cache file (%llu) is invalid, skip it", - offset); - return 0; - } - - if (BTRFS_I(inode)->generation != generation) { - btrfs_err(fs_info, - "free space inode generation (%llu) did not match free space cache generation (%llu)", - BTRFS_I(inode)->generation, generation); - return 0; - } - - if (!num_entries) - return 0; - - ret = io_ctl_init(&io_ctl, inode, 0); - if (ret) - return ret; - - readahead_cache(inode); - - ret = io_ctl_prepare_pages(&io_ctl, true); - if (ret) - goto out; - - ret = io_ctl_check_crc(&io_ctl, 0); - if (ret) - goto free_cache; - - ret = io_ctl_check_generation(&io_ctl, generation); - if (ret) - goto free_cache; - - while (num_entries) { - e = kmem_cache_zalloc(btrfs_free_space_cachep, - GFP_NOFS); - if (!e) { - ret = -ENOMEM; - goto free_cache; - } - - ret = io_ctl_read_entry(&io_ctl, e, &type); - if (ret) { - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (!e->bytes) { - ret = -1; - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (type == BTRFS_FREE_SPACE_EXTENT) { - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - spin_unlock(&ctl->tree_lock); - if (ret) { - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - } else { - ASSERT(num_bitmaps); - num_bitmaps--; - e->bitmap = kmem_cache_zalloc( - btrfs_free_space_bitmap_cachep, GFP_NOFS); - if (!e->bitmap) { - ret = -ENOMEM; - kmem_cache_free( - btrfs_free_space_cachep, e); - goto free_cache; - } - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - if (ret) { - spin_unlock(&ctl->tree_lock); - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_bitmap_cachep, e->bitmap); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - ctl->total_bitmaps++; - recalculate_thresholds(ctl); - spin_unlock(&ctl->tree_lock); - list_add_tail(&e->list, &bitmaps); - } - - num_entries--; - } - - io_ctl_unmap_page(&io_ctl); - - /* - * We add the bitmaps at the end of the entries in order that - * the bitmap entries are added to the cache. - */ - list_for_each_entry_safe(e, n, &bitmaps, list) { - list_del_init(&e->list); - ret = io_ctl_read_bitmap(&io_ctl, e); - if (ret) - goto free_cache; - } - - io_ctl_drop_pages(&io_ctl); - ret = 1; -out: - io_ctl_free(&io_ctl); - return ret; -free_cache: - io_ctl_drop_pages(&io_ctl); - - spin_lock(&ctl->tree_lock); - __btrfs_remove_free_space_cache(ctl); - spin_unlock(&ctl->tree_lock); - goto out; -} - -static int copy_free_space_cache(struct btrfs_free_space_ctl *ctl) -{ - struct btrfs_free_space *info; - struct rb_node *n; - int ret = 0; - - while (!ret && (n = rb_first(&ctl->free_space_offset)) != NULL) { - info = rb_entry(n, struct btrfs_free_space, offset_index); - if (!info->bitmap) { - const u64 offset = info->offset; - const u64 bytes = info->bytes; - - unlink_free_space(ctl, info, true); - spin_unlock(&ctl->tree_lock); - kmem_cache_free(btrfs_free_space_cachep, info); - ret = btrfs_add_free_space(ctl->block_group, offset, bytes); - spin_lock(&ctl->tree_lock); - } else { - u64 offset = info->offset; - u64 bytes = ctl->block_group->fs_info->sectorsize; - - ret = search_bitmap(ctl, info, &offset, &bytes, false); - if (ret == 0) { - bitmap_clear_bits(ctl, info, offset, bytes, true); - spin_unlock(&ctl->tree_lock); - ret = btrfs_add_free_space(ctl->block_group, offset, - bytes); - spin_lock(&ctl->tree_lock); - } else { - free_bitmap(ctl, info); - ret = 0; - } - } - cond_resched_lock(&ctl->tree_lock); - } - return ret; -} - -static struct lock_class_key btrfs_free_space_inode_key; - -int load_free_space_cache(struct btrfs_block_group *block_group) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_space_ctl tmp_ctl = {}; - struct inode *inode; - struct btrfs_path *path; - int ret = 0; - bool matched; - u64 used = block_group->used; - - /* - * Because we could potentially discard our loaded free space, we want - * to load everything into a temporary structure first, and then if it's - * valid copy it all into the actual free space ctl. - */ - btrfs_init_free_space_ctl(block_group, &tmp_ctl); - - /* - * If this block group has been marked to be cleared for one reason or - * another then we can't trust the on disk cache, so just return. - */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - path = btrfs_alloc_path(); - if (!path) - return 0; - path->search_commit_root = true; - path->skip_locking = true; - - /* - * We must pass a path with search_commit_root set to btrfs_iget in - * order to avoid a deadlock when allocating extents for the tree root. - * - * When we are COWing an extent buffer from the tree root, when looking - * for a free extent, at extent-tree.c:find_free_extent(), we can find - * block group without its free space cache loaded. When we find one - * we must load its space cache which requires reading its free space - * cache's inode item from the root tree. If this inode item is located - * in the same leaf that we started COWing before, then we end up in - * deadlock on the extent buffer (trying to read lock it when we - * previously write locked it). - * - * It's safe to read the inode item using the commit root because - * block groups, once loaded, stay in memory forever (until they are - * removed) as well as their space caches once loaded. New block groups - * once created get their ->cached field set to BTRFS_CACHE_FINISHED so - * we will never try to read their inode item while the fs is mounted. - */ - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) { - btrfs_free_path(path); - return 0; - } - - /* We may have converted the inode and made the cache invalid. */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - btrfs_free_path(path); - goto out; - } - spin_unlock(&block_group->lock); - - /* - * Reinitialize the class of struct inode's mapping->invalidate_lock for - * free space inodes to prevent false positives related to locks for normal - * inodes. - */ - lockdep_set_class(&(&inode->i_data)->invalidate_lock, - &btrfs_free_space_inode_key); - - ret = __load_free_space_cache(fs_info->tree_root, inode, &tmp_ctl, - path, block_group->start); - btrfs_free_path(path); - if (ret <= 0) - goto out; - - matched = (tmp_ctl.free_space == (block_group->length - used - - block_group->bytes_super)); - - if (matched) { - spin_lock(&tmp_ctl.tree_lock); - ret = copy_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - /* - * ret == 1 means we successfully loaded the free space cache, - * so we need to re-set it here. - */ - if (ret == 0) - ret = 1; - } else { - /* - * We need to call the _locked variant so we don't try to update - * the discard counters. - */ - spin_lock(&tmp_ctl.tree_lock); - __btrfs_remove_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - btrfs_warn(fs_info, - "block group %llu has wrong amount of free space", - block_group->start); - ret = -1; - } -out: - if (ret < 0) { - /* This cache is bogus, make sure it gets cleared */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - ret = 0; - - btrfs_warn(fs_info, - "failed to load free space cache for block group %llu, rebuilding it now", - block_group->start); - } - - spin_lock(&ctl->tree_lock); - btrfs_discard_update_discardable(block_group); - spin_unlock(&ctl->tree_lock); - iput(inode); - return ret; -} - -static noinline_for_stack -int write_cache_extent_entries(struct btrfs_io_ctl *io_ctl, - struct btrfs_block_group *block_group, - int *entries, int *bitmaps, - struct list_head *bitmap_list) -{ - int ret; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_cluster *cluster = NULL; - struct btrfs_free_cluster *cluster_locked = NULL; - struct rb_node *node = rb_first(&ctl->free_space_offset); - struct btrfs_trim_range *trim_entry; - - /* Get the cluster for this block_group if it exists */ - if (!list_empty(&block_group->cluster_list)) { - cluster = list_first_entry(&block_group->cluster_list, - struct btrfs_free_cluster, block_group_list); - } - - if (!node && cluster) { - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - node = rb_first(&cluster->root); - cluster = NULL; - } - - /* Write out the extent entries */ - while (node) { - struct btrfs_free_space *e; - - e = rb_entry(node, struct btrfs_free_space, offset_index); - *entries += 1; - - ret = io_ctl_add_entry(io_ctl, e->offset, e->bytes, - e->bitmap); - if (ret) - goto fail; - - if (e->bitmap) { - list_add_tail(&e->list, bitmap_list); - *bitmaps += 1; - } - node = rb_next(node); - if (!node && cluster) { - node = rb_first(&cluster->root); - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - cluster = NULL; - } - } - if (cluster_locked) { - spin_unlock(&cluster_locked->lock); - cluster_locked = NULL; - } - - /* - * Make sure we don't miss any range that was removed from our rbtree - * because trimming is running. Otherwise after a umount+mount (or crash - * after committing the transaction) we would leak free space and get - * an inconsistent free space cache report from fsck. - */ - list_for_each_entry(trim_entry, &ctl->trimming_ranges, list) { - ret = io_ctl_add_entry(io_ctl, trim_entry->start, - trim_entry->bytes, NULL); - if (ret) - goto fail; - *entries += 1; - } - - return 0; -fail: - if (cluster_locked) - spin_unlock(&cluster_locked->lock); - return -ENOSPC; -} - -static noinline_for_stack int -update_cache_item(struct btrfs_trans_handle *trans, - struct btrfs_root *root, - struct inode *inode, - struct btrfs_path *path, u64 offset, - int entries, int bitmaps) -{ - struct btrfs_key key; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - int ret; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(trans, root, &key, path, 0, 1); - if (ret < 0) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - return ret; - } - leaf = path->nodes[0]; - if (ret > 0) { - struct btrfs_key found_key; - ASSERT(path->slots[0]); - path->slots[0]--; - btrfs_item_key_to_cpu(leaf, &found_key, path->slots[0]); - if (found_key.objectid != BTRFS_FREE_SPACE_OBJECTID || - found_key.offset != offset) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, - inode->i_size - 1, EXTENT_DELALLOC, - NULL); - btrfs_release_path(path); - return -ENOENT; - } - } - - BTRFS_I(inode)->generation = trans->transid; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - btrfs_set_free_space_entries(leaf, header, entries); - btrfs_set_free_space_bitmaps(leaf, header, bitmaps); - btrfs_set_free_space_generation(leaf, header, trans->transid); - btrfs_release_path(path); - - return 0; -} - -static noinline_for_stack int write_pinned_extent_entries( - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - int *entries) -{ - u64 start, extent_start, extent_end, len; - const u64 block_group_end = btrfs_block_group_end(block_group); - struct extent_io_tree *unpin = NULL; - int ret; - - /* - * We want to add any pinned extents to our free space cache - * so we don't leak the space - * - * We shouldn't have switched the pinned extents yet so this is the - * right one - */ - unpin = &trans->transaction->pinned_extents; - - start = block_group->start; - - while (start < block_group_end) { - if (!btrfs_find_first_extent_bit(unpin, start, - &extent_start, &extent_end, - EXTENT_DIRTY, NULL)) - return 0; - - /* This pinned extent is out of our range */ - if (extent_start >= block_group_end) - return 0; - - extent_start = max(extent_start, start); - extent_end = min(block_group_end, extent_end + 1); - len = extent_end - extent_start; - - *entries += 1; - ret = io_ctl_add_entry(io_ctl, extent_start, len, NULL); - if (ret) - return -ENOSPC; - - start = extent_end; - } - - return 0; -} - -static noinline_for_stack int -write_bitmap_entries(struct btrfs_io_ctl *io_ctl, struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - int ret; - - /* Write out the bitmaps */ - list_for_each_entry_safe(entry, next, bitmap_list, list) { - ret = io_ctl_add_bitmap(io_ctl, entry->bitmap); - if (ret) - return -ENOSPC; - list_del_init(&entry->list); - } - - return 0; -} - -static int flush_dirty_cache(struct inode *inode) -{ - int ret; - - ret = btrfs_wait_ordered_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - - return ret; -} - -static void noinline_for_stack -cleanup_bitmap_list(struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - - list_for_each_entry_safe(entry, next, bitmap_list, list) - list_del_init(&entry->list); -} - -static void noinline_for_stack -cleanup_write_cache_enospc(struct inode *inode, - struct btrfs_io_ctl *io_ctl, - struct extent_state **cached_state) -{ - io_ctl_drop_pages(io_ctl); - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - cached_state); -} - -static int __btrfs_wait_cache_io(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - struct btrfs_path *path, u64 offset) -{ - int ret; - struct inode *inode = io_ctl->inode; - - if (!inode) - return 0; - - /* Flush the dirty pages in the cache file. */ - ret = flush_dirty_cache(inode); - if (ret) - goto out; - - /* Update the cache item to tell everyone this cache file is valid. */ - ret = update_cache_item(trans, root, inode, path, offset, - io_ctl->entries, io_ctl->bitmaps); -out: - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - if (block_group) - btrfs_debug(root->fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - } - btrfs_update_inode(trans, BTRFS_I(inode)); - - if (block_group) { - /* the dirty list is protected by the dirty_bgs_lock */ - spin_lock(&trans->transaction->dirty_bgs_lock); - - /* the disk_cache_state is protected by the block group lock */ - spin_lock(&block_group->lock); - - /* - * only mark this as written if we didn't get put back on - * the dirty list while waiting for IO. Otherwise our - * cache state won't be right, and we won't get written again - */ - if (!ret && list_empty(&block_group->dirty_list)) - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - else if (ret) - block_group->disk_cache_state = BTRFS_DC_ERROR; - - spin_unlock(&block_group->lock); - spin_unlock(&trans->transaction->dirty_bgs_lock); - io_ctl->inode = NULL; - iput(inode); - } - - return ret; - -} - -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - return __btrfs_wait_cache_io(block_group->fs_info->tree_root, trans, - block_group, &block_group->io_ctl, - path, block_group->start); -} - -/* - * Write out cached info to an inode. - * - * @inode: freespace inode we are writing out - * @ctl: free space cache we are going to write out - * @block_group: block_group for this cache if it belongs to a block_group - * @io_ctl: holds context for the io - * @trans: the trans handle - * - * This function writes out a free space cache struct to disk for quick recovery - * on mount. This will return 0 if it was successful in writing the cache out, - * or an errno if it was not. - */ -static int __btrfs_write_out_cache(struct inode *inode, - struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans) -{ - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_io_ctl *io_ctl = &block_group->io_ctl; - struct extent_state *cached_state = NULL; - LIST_HEAD(bitmap_list); - int entries = 0; - int bitmaps = 0; - int ret; - bool must_iput = false; - int i_size; - - if (!i_size_read(inode)) - return -EIO; - - WARN_ON(io_ctl->pages); - ret = io_ctl_init(io_ctl, inode, 1); - if (ret) - return ret; - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) { - down_write(&block_group->data_rwsem); - spin_lock(&block_group->lock); - if (block_group->delalloc_bytes) { - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - up_write(&block_group->data_rwsem); - BTRFS_I(inode)->generation = 0; - ret = 0; - must_iput = true; - goto out; - } - spin_unlock(&block_group->lock); - } - - /* Lock all pages first so we can lock the extent safely. */ - ret = io_ctl_prepare_pages(io_ctl, false); - if (ret) - goto out_unlock; - - btrfs_lock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - io_ctl_set_generation(io_ctl, trans->transid); - - mutex_lock(&ctl->cache_writeout_mutex); - /* Write out the extent entries in the free space cache */ - spin_lock(&ctl->tree_lock); - ret = write_cache_extent_entries(io_ctl, block_group, &entries, &bitmaps, - &bitmap_list); - if (ret) - goto out_nospc_locked; - - /* - * Some spaces that are freed in the current transaction are pinned, - * they will be added into free space cache after the transaction is - * committed, we shouldn't lose them. - * - * If this changes while we are working we'll get added back to - * the dirty list and redo it. No locking needed - */ - ret = write_pinned_extent_entries(trans, block_group, io_ctl, &entries); - if (ret) - goto out_nospc_locked; - - /* - * At last, we write out all the bitmaps and keep cache_writeout_mutex - * locked while doing it because a concurrent trim can be manipulating - * or freeing the bitmap. - */ - ret = write_bitmap_entries(io_ctl, &bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - if (ret) - goto out_nospc; - - /* Zero out the rest of the pages just to make sure */ - io_ctl_zero_remaining_pages(io_ctl); - - /* Everything is written out, now we dirty the pages in the file. */ - i_size = i_size_read(inode); - for (int i = 0; i < round_up(i_size, PAGE_SIZE) / PAGE_SIZE; i++) { - u64 dirty_start = i * PAGE_SIZE; - u64 dirty_len = min_t(u64, dirty_start + PAGE_SIZE, i_size) - dirty_start; - - ret = btrfs_dirty_folio(BTRFS_I(inode), page_folio(io_ctl->pages[i]), - dirty_start, dirty_len, &cached_state, false); - if (ret < 0) - goto out_nospc; - } - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - /* - * Release the pages and unlock the extent, we will flush - * them out later - */ - io_ctl_drop_pages(io_ctl); - io_ctl_free(io_ctl); - - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - /* - * at this point the pages are under IO and we're happy, - * The caller is responsible for waiting on them and updating - * the cache and the inode - */ - io_ctl->entries = entries; - io_ctl->bitmaps = bitmaps; - - ret = btrfs_fdatawrite_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - goto out; - - return 0; - -out_nospc_locked: - cleanup_bitmap_list(&bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - -out_nospc: - cleanup_write_cache_enospc(inode, io_ctl, &cached_state); - -out_unlock: - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - -out: - io_ctl->inode = NULL; - io_ctl_free(io_ctl); - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - } - btrfs_update_inode(trans, BTRFS_I(inode)); - if (must_iput) - iput(inode); - return ret; -} - -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct inode *inode; - int ret = 0; - - spin_lock(&block_group->lock); - if (block_group->disk_cache_state < BTRFS_DC_SETUP) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) - return 0; - - ret = __btrfs_write_out_cache(inode, block_group, trans); - if (ret) { - btrfs_debug(fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&block_group->lock); - - block_group->io_ctl.inode = NULL; - iput(inode); - } - - /* - * if ret == 0 the caller is expected to call btrfs_wait_cache_io - * to wait for IO and put the inode - */ - - return ret; -} - static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit, u64 offset) { @@ -2953,8 +1686,6 @@ void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, spin_lock_init(&ctl->tree_lock); ctl->block_group = block_group; ctl->free_space_bytes = RB_ROOT_CACHED; - INIT_LIST_HEAD(&ctl->trimming_ranges); - mutex_init(&ctl->cache_writeout_mutex); /* * we only want to have 32k of ram per block group for keeping @@ -3650,12 +2381,10 @@ void btrfs_init_free_cluster(struct btrfs_free_cluster *cluster) static int do_trimming(struct btrfs_block_group *block_group, u64 *total_trimmed, u64 start, u64 bytes, u64 reserved_start, u64 reserved_bytes, - enum btrfs_trim_state reserved_trim_state, - struct btrfs_trim_range *trim_entry) + enum btrfs_trim_state reserved_trim_state) { struct btrfs_space_info *space_info = block_group->space_info; struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; int ret; bool bg_ro; const u64 end = start + bytes; @@ -3681,7 +2410,6 @@ static int do_trimming(struct btrfs_block_group *block_group, trim_state = BTRFS_TRIM_STATE_TRIMMED; } - mutex_lock(&ctl->cache_writeout_mutex); if (reserved_start < start) __btrfs_add_free_space(block_group, reserved_start, start - reserved_start, @@ -3690,8 +2418,6 @@ static int do_trimming(struct btrfs_block_group *block_group, __btrfs_add_free_space(block_group, end, reserved_end - end, reserved_trim_state); __btrfs_add_free_space(block_group, start, bytes, trim_state); - list_del(&trim_entry->list); - mutex_unlock(&ctl->cache_writeout_mutex); if (!bg_ro) { spin_lock(&space_info->lock); @@ -3729,9 +2455,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, const u64 max_discard_size = READ_ONCE(discard_ctl->max_discard_size); while (start < end) { - struct btrfs_trim_range trim_entry; - - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) @@ -3762,7 +2485,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = entry->bytes; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } unlink_free_space(ctl, entry, true); @@ -3787,7 +2509,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = min(extent_start + extent_bytes, end) - start; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -3796,14 +2517,9 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, } spin_unlock(&ctl->tree_lock); - trim_entry.start = extent_start; - trim_entry.bytes = extent_bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - extent_start, extent_bytes, extent_trim_state, - &trim_entry); + extent_start, extent_bytes, extent_trim_state); if (ret) { block_group->discard_cursor = start + bytes; break; @@ -3827,7 +2543,6 @@ next: out_unlock: block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } @@ -3938,16 +2653,13 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, while (offset < end) { bool next_bitmap = false; - struct btrfs_trim_range trim_entry; - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) { block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); break; } @@ -3963,7 +2675,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, if (!entry || (async && minlen && start == offset && btrfs_free_space_trimmed(entry))) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -3989,7 +2700,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, else entry->trim_state = BTRFS_TRIM_STATE_UNTRIMMED; spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -4000,14 +2710,12 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, */ if (async && *total_trimmed) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } bytes = min(bytes, end - start); if (bytes < minlen || (async && maxlen && bytes > maxlen)) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -4027,13 +2735,9 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, free_bitmap(ctl, entry); spin_unlock(&ctl->tree_lock); - trim_entry.start = start; - trim_entry.bytes = bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - start, bytes, 0, &trim_entry); + start, bytes, 0); if (ret) { reset_trimming_bitmap(ctl, offset); block_group->discard_cursor = @@ -4152,47 +2856,29 @@ bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info) return btrfs_super_cache_generation(fs_info->super_copy); } -static int cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info, - struct btrfs_trans_handle *trans) +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info) { - struct btrfs_block_group *block_group; + struct btrfs_trans_handle *trans; struct rb_node *node; + int ret; btrfs_info(fs_info, "cleaning free space cache v1"); - node = rb_first_cached(&fs_info->block_group_cache_tree); - while (node) { - int ret; - - block_group = rb_entry(node, struct btrfs_block_group, cache_node); - ret = btrfs_remove_free_space_inode(trans, NULL, block_group); - if (ret) - return ret; - node = rb_next(node); - } - return 0; -} - -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active) -{ - struct btrfs_trans_handle *trans; - int ret; - /* - * update_super_roots will appropriately set or unset - * super_copy->cache_generation based on SPACE_CACHE and - * BTRFS_FS_CLEANUP_SPACE_CACHE_V1. For this reason, we need a - * transaction commit whether we are enabling space cache v1 and don't - * have any other work to do, or are disabling it and removing free - * space inodes. + * update_super_roots() zeroes super_copy->cache_generation while + * BTRFS_FS_CLEANUP_SPACE_CACHE_V1 is set, so this needs a commit. */ trans = btrfs_start_transaction(fs_info->tree_root, 0); if (IS_ERR(trans)) return PTR_ERR(trans); - if (!active) { - set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); - ret = cleanup_free_space_cache_v1(fs_info, trans); + set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); + for (node = rb_first_cached(&fs_info->block_group_cache_tree); node; + node = rb_next(node)) { + struct btrfs_block_group *block_group; + + block_group = rb_entry(node, struct btrfs_block_group, cache_node); + ret = btrfs_remove_free_space_inode(trans, NULL, block_group); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 53fe8e293af1..e22443598b8e 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -14,7 +14,6 @@ #include "fs.h" struct inode; -struct page; struct btrfs_fs_info; struct btrfs_path; struct btrfs_trans_handle; @@ -84,44 +83,18 @@ struct btrfs_free_space_ctl { s32 discardable_extents[BTRFS_STAT_NR_ENTRIES]; s64 discardable_bytes[BTRFS_STAT_NR_ENTRIES]; struct btrfs_block_group *block_group; - struct mutex cache_writeout_mutex; - struct list_head trimming_ranges; -}; - -struct btrfs_io_ctl { - void *cur, *orig; - struct page *page; - struct page **pages; - struct btrfs_fs_info *fs_info; - struct inode *inode; - unsigned long size; - int index; - int num_pages; - int entries; - int bitmaps; }; int __init btrfs_free_space_init(void); void __cold btrfs_free_space_exit(void); struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, struct btrfs_path *path); -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, struct inode *inode, struct btrfs_block_group *block_group); int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *inode); -int load_free_space_cache(struct btrfs_block_group *block_group); -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, struct btrfs_free_space_ctl *ctl); @@ -161,7 +134,7 @@ int btrfs_trim_block_group_bitmaps(struct btrfs_block_group *block_group, void btrfs_trim_fully_remapped_block_group(struct btrfs_block_group *bg); bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info); -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active); +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info); /* Support functions for running our sanity tests */ #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS bool btrfs_use_bitmap(struct btrfs_free_space_ctl *ctl, diff --git a/fs/btrfs/fs.c b/fs/btrfs/fs.c index de160d29dde8..75a1217727a7 100644 --- a/fs/btrfs/fs.c +++ b/fs/btrfs/fs.c @@ -79,7 +79,7 @@ void btrfs_csum_init(struct btrfs_csum_ctx *ctx, u16 csum_type) blake2b_init(&ctx->blake2b, 32); break; default: - /* Checksume type is validated at mount time. */ + /* Checksum type is validated at mount time. */ BUG(); } } diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 10e15a319b93..79d0828c51c7 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -259,7 +259,6 @@ enum { BTRFS_MOUNT_NOSSD = (1ULL << 9), BTRFS_MOUNT_DISCARD_SYNC = (1ULL << 10), BTRFS_MOUNT_FORCE_COMPRESS = (1ULL << 11), - BTRFS_MOUNT_SPACE_CACHE = (1ULL << 12), BTRFS_MOUNT_CLEAR_CACHE = (1ULL << 13), BTRFS_MOUNT_USER_SUBVOL_RM_ALLOWED = (1ULL << 14), BTRFS_MOUNT_ENOSPC_DEBUG = (1ULL << 15), @@ -712,7 +711,6 @@ struct btrfs_fs_info { struct workqueue_struct *endio_meta_workers; struct workqueue_struct *rmw_workers; struct btrfs_workqueue *endio_write_workers; - struct btrfs_workqueue *endio_freespace_worker; struct btrfs_workqueue *caching_workers; struct workqueue_struct *fixup_workers; @@ -811,7 +809,7 @@ struct btrfs_fs_info { struct btrfs_discard_ctl discard_ctl; /* Is qgroup tracking in a consistent state? */ - u64 qgroup_flags; + unsigned long qgroup_flags; /* Holds configuration and tracking. Protected by qgroup_lock. */ struct rb_root qgroup_tree; diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 558b4a3f9633..f4b68205f621 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -730,6 +730,9 @@ static inline int inode_need_compress(struct btrfs_inode *inode, u64 start, u64 end, bool check_inline) { struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; + + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); if (unlikely(!btrfs_inode_can_compress(inode))) { DEBUG_WARN("BTRFS: unexpected compression for ino %llu", btrfs_ino(inode)); @@ -1366,11 +1369,6 @@ static noinline int cow_file_range(struct btrfs_inode *inode, goto out_unlock; } - if (btrfs_is_free_space_inode(inode)) { - ret = -EINVAL; - goto out_unlock; - } - num_bytes = ALIGN(end - start + 1, blocksize); num_bytes = max(blocksize, num_bytes); ASSERT(num_bytes <= btrfs_super_total_bytes(fs_info->super_copy)); @@ -1680,7 +1678,6 @@ static int fallback_to_cow(struct btrfs_inode *inode, struct folio *locked_folio, const u64 start, const u64 end) { - const bool is_space_ino = btrfs_is_free_space_inode(inode); const bool is_reloc_ino = btrfs_is_data_reloc_root(inode->root); const u64 range_bytes = end + 1 - start; struct extent_io_tree *io_tree = &inode->io_tree; @@ -1713,23 +1710,22 @@ static int fallback_to_cow(struct btrfs_inode *inode, * extent_clear_unlock_delalloc()) the bytes_may_use counter of the * data space info, which we incremented in the step above. * - * If we need to fallback to cow and the inode corresponds to a free - * space cache inode or an inode of the data relocation tree, we must - * also increment bytes_may_use of the data space_info for the same - * reason. Space caches and relocated data extents always get a prealloc - * extent for them, however scrub or balance may have set the block - * group that contains that extent to RO mode and therefore force COW - * when starting writeback. + * If we need to fallback to cow and the inode is in the data relocation + * tree, we must also increment bytes_may_use of the data space_info for + * the same reason. Relocated data extents always get a prealloc extent, + * however scrub or balance may have set the block group that contains + * that extent to RO mode and therefore force COW when starting + * writeback. */ btrfs_lock_extent(io_tree, start, end, &cached_state); count = btrfs_count_range_bits(io_tree, &range_start, end, range_bytes, EXTENT_NORESERVE, false, NULL); - if (count > 0 || is_space_ino || is_reloc_ino) { + if (count > 0 || is_reloc_ino) { u64 bytes = count; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_space_info *sinfo = fs_info->data_sinfo; - if (is_space_ino || is_reloc_ino) + if (is_reloc_ino) bytes = range_bytes; spin_lock(&sinfo->lock); @@ -1794,7 +1790,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, struct btrfs_inode *inode, struct can_nocow_file_extent_args *args) { - const bool is_freespace_inode = btrfs_is_free_space_inode(inode); struct extent_buffer *leaf = path->nodes[0]; struct btrfs_root *root = inode->root; struct btrfs_file_extent_item *fi; @@ -1807,8 +1802,7 @@ static int can_nocow_file_extent(struct btrfs_path *path, bool nowait = path->nowait; /* If there are pending snapshots for this root, we must do COW. */ - if (args->writeback_path && !is_freespace_inode && - atomic_read(&root->snapshot_force_cow)) + if (args->writeback_path && atomic_read(&root->snapshot_force_cow)) goto out; fi = btrfs_item_ptr(leaf, path->slots[0], struct btrfs_file_extent_item); @@ -1857,7 +1851,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_cross_ref_exist(inode, key->offset - args->file_extent.offset, args->file_extent.disk_bytenr, path); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -1892,7 +1885,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_lookup_csums_list(csum_root, io_start, io_start + args->file_extent.num_bytes - 1, NULL, nowait); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -2331,7 +2323,7 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f btrfs_check_folio_write_protected(locked_folio); if (btrfs_inode_can_compress(inode) && - inode_need_compress(inode, 0, blocksize, true)) { + inode_need_compress(inode, 0, blocksize - 1, true)) { if (inode->defrag_compress > 0 && inode->defrag_compress < BTRFS_NR_COMPRESS_TYPES) { compress_type = inode->defrag_compress; @@ -2640,7 +2632,7 @@ void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *s * and are therefore protected against concurrent calls of this * function and btrfs_clear_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && prev_delalloc_bytes == 0) + if (prev_delalloc_bytes == 0) btrfs_add_delalloc_inode(inode); } @@ -2698,7 +2690,6 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, return; if (!btrfs_is_data_reloc_root(root) && - !btrfs_is_free_space_inode(inode) && !(state->state & EXTENT_NORESERVE) && (bits & EXTENT_CLEAR_DATA_RESV)) btrfs_free_reserved_data_space_noquota(inode, len); @@ -2716,7 +2707,7 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, * and are therefore protected against concurrent calls of this * function and btrfs_set_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && new_delalloc_bytes == 0) { + if (new_delalloc_bytes == 0) { spin_lock(&root->delalloc_lock); btrfs_del_delalloc_inode(inode); spin_unlock(&root->delalloc_lock); @@ -3218,7 +3209,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) int compress_type = 0; int ret = 0; u64 logical_len = ordered_extent->num_bytes; - bool freespace_inode; + u64 unwritten_start; bool truncated = false; bool clear_reserved_extent = true; unsigned int clear_bits = 0; @@ -3235,9 +3226,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) if (!test_bit(BTRFS_ORDERED_NOCOW, &ordered_extent->flags)) clear_bits |= EXTENT_DEFRAG; - freespace_inode = btrfs_is_free_space_inode(inode); - if (!freespace_inode) - btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); if (unlikely(test_bit(BTRFS_ORDERED_IOERR, &ordered_extent->flags))) { ret = -EIO; @@ -3272,10 +3261,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) &cached_state); } - if (freespace_inode) - trans = btrfs_join_transaction_spacecache(root); - else - trans = btrfs_join_transaction(root); + trans = btrfs_join_transaction(root); if (IS_ERR(trans)) { ret = PTR_ERR(trans); trans = NULL; @@ -3385,29 +3371,11 @@ out: if (ret) btrfs_mark_ordered_extent_error(ordered_extent); - /* - * Drop extent maps for the part of the extent we didn't write. - * - * We have an exception here for the free_space_inode, this is - * because when we do btrfs_get_extent() on the free space inode - * we will search the commit root. If this is a new block group - * we won't find anything, and we will trip over the assert in - * writepage where we do ASSERT(em->block_start != - * EXTENT_MAP_HOLE). - * - * Theoretically we could also skip this for any NOCOW extent as - * we don't mess with the extent map tree in the NOCOW case, but - * for now simply skip this if we are the free space inode. - */ - if (!btrfs_is_free_space_inode(inode)) { - u64 unwritten_start = start; - - if (truncated) - unwritten_start += logical_len; - - btrfs_drop_extent_map_range(inode, unwritten_start, - end, false); - } + /* Drop extent maps for the part of the extent we didn't write. */ + unwritten_start = start; + if (truncated) + unwritten_start += logical_len; + btrfs_drop_extent_map_range(inode, unwritten_start, end, false); /* * If the ordered extent had an IOERR or something else went @@ -3471,79 +3439,30 @@ int btrfs_finish_ordered_io(struct btrfs_ordered_extent *ordered) return btrfs_finish_one_ordered(ordered); } -/* - * Calculate the checksum of an fs block at physical memory address @paddr, - * and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest) +/* Generate data checksum for a single fs block, pointed to by @orig_iter. */ +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum) { - struct folio *folio = page_folio(phys_to_page(paddr)); + struct btrfs_csum_ctx cctx; + struct bvec_iter iter = *orig_iter; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - - /* The full block must be inside the folio. */ - ASSERT(offset_in_folio(folio, paddr) + blocksize <= folio_size(folio)); + u32 cur = 0; - for (int i = 0; i < nr_steps; i++) { - u32 pindex = offset_in_folio(folio, paddr + i * step) >> PAGE_SHIFT; - - /* - * For bs <= ps cases, we will only run the loop once, so the offset - * inside the page will only added to paddrs[0]. - * - * For bs > ps cases, the block must be page aligned, thus offset - * inside the page will always be 0. - */ - paddrs[i] = page_to_phys(folio_page(folio, pindex)) + offset_in_page(paddr); - } - return btrfs_calculate_block_csum_pages(fs_info, paddrs, dest); -} - -/* - * Calculate the checksum of a fs block backed by multiple noncontiguous pages - * at @paddrs[] and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest) -{ - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - struct btrfs_csum_ctx csum; - - btrfs_csum_init(&csum, fs_info->csum_type); - for (int i = 0; i < nr_steps; i++) { - const phys_addr_t paddr = paddrs[i]; + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(bio, iter); + const u32 pg_off = bio_iter_offset(bio, iter); + const u32 cur_len = min(bio_iter_len(bio, iter), blocksize - cur); void *kaddr; - ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); - kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); - btrfs_csum_update(&csum, kaddr, step); + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); kunmap_local(kaddr); - } - btrfs_csum_final(&csum, dest); -} -/* - * Verify the checksum for a single sector without any extra action that depend - * on the type of I/O. - * - * @kaddr must be a properly kmapped address. - */ -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected) -{ - btrfs_calculate_block_csum_folio(fs_info, paddr, csum); - if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) - return -EIO; - return 0; + bio_advance_iter_single(bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); } /* @@ -3551,27 +3470,30 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 * different noncontiguous pages. * * @bbio: btrfs_io_bio which contains the csum - * @dev: device the sector is on - * @bio_offset: offset to the beginning of the bio (in bytes) - * @paddrs: physical addresses which back the fs block + * @orig_iter: bvec iter pointing to the start of the block + * @dev: device the sector is on (optional) * * Check if the checksum on a data block is valid. When a checksum mismatch is * detected, report the error and fill the corrupted range with zero. * * Return %true if the sector is ok or had no checksum to start with, else %false. */ -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]) +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, + const struct bvec_iter *orig_iter, + struct btrfs_device *dev) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; + struct bvec_iter iter = *orig_iter; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; + const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; u64 end = file_offset + blocksize - 1; u8 *csum_expected; u8 csum[BTRFS_CSUM_SIZE]; + u32 cur = 0; + + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); if (!bbio->csum) return true; @@ -3587,7 +3509,7 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum); + btrfs_csum_one_bio_block(fs_info, &bbio->bio, orig_iter, csum); if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3597,8 +3519,16 @@ zeroit: bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - for (int i = 0; i < nr_steps; i++) - memzero_page(phys_to_page(paddrs[i]), offset_in_page(paddrs[i]), step); + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + + memzero_page(page, pg_off, cur_len); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } return false; } @@ -6881,8 +6811,28 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, } } else { ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name, - false, BTRFS_I(inode)->dir_index); - if (unlikely(ret)) { + false, BTRFS_I(inode)->dir_index, NULL); + if (ret == -ENOMEM) { + /* + * Orphan the new inode instead of aborting. The inode + * item was already written with nlink 1, and discard's + * eviction won't delete a bad inode, so nlink 0 must be + * persisted here or orphan cleanup would see nlink > 0, + * drop the orphan item, and leak the inode. + */ + clear_nlink(inode); + /* btrfs_orphan_add() aborts the transaction on failure. */ + ret = btrfs_orphan_add(trans, BTRFS_I(inode)); + if (ret) + goto discard; + ret = btrfs_update_inode(trans, BTRFS_I(inode)); + if (ret) { + btrfs_abort_transaction(trans, ret); + goto discard; + } + ret = -ENOMEM; + goto discard; + } else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto discard; } @@ -6913,7 +6863,8 @@ out: */ int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index) + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; struct btrfs_key key; @@ -6939,12 +6890,14 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, } /* Nothing to clean up yet */ - if (ret) + if (ret) { + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); return ret; + } ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, - btrfs_inode_type(inode), index); - if (ret == -EEXIST || ret == -EOVERFLOW) + btrfs_inode_type(inode), index, prealloc); + if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM) goto fail_dir_item; else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); @@ -7097,7 +7050,7 @@ static int btrfs_link(struct dentry *old_dentry, struct inode *dir, inode_set_ctime_current(inode); ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), - &fname.disk_name, true, index); + &fname.disk_name, true, index, NULL); if (ret) goto fail; @@ -7164,7 +7117,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, compress_type = btrfs_file_extent_compression(leaf, item); max_size = btrfs_file_extent_ram_bytes(leaf, item); inline_size = btrfs_file_extent_inline_item_len(leaf, path->slots[0]); - tmp = kmalloc(inline_size, GFP_NOFS); + tmp = kvmalloc(inline_size, GFP_NOFS); if (!tmp) return -ENOMEM; ptr = btrfs_file_extent_inline_start(item); @@ -7185,7 +7138,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, if (max_size < blocksize) folio_zero_range(folio, max_size, blocksize - max_size); - kfree(tmp); + kvfree(tmp); return ret; } @@ -7281,16 +7234,6 @@ struct extent_map *btrfs_get_extent(struct btrfs_inode *inode, /* Chances are we'll be called again, so go ahead and do readahead */ path->reada = READA_FORWARD; - /* - * The same explanation in load_free_space_cache applies here as well, - * we only read when we're loading the free space cache, and at that - * point the commit_root has everything we need. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - ret = btrfs_lookup_file_extent(NULL, root, path, objectid, start, 0); if (ret < 0) { goto out; @@ -8147,7 +8090,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) struct btrfs_ordered_extent *ordered; struct btrfs_inode *inode = BTRFS_I(vfs_inode); struct btrfs_root *root = inode->root; - bool freespace_inode; WARN_ON(!hlist_empty(&vfs_inode->i_dentry)); WARN_ON(vfs_inode->i_data.nrpages); @@ -8170,12 +8112,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) if (!root) return; - /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - while (1) { ordered = btrfs_lookup_first_ordered_extent(inode, (u64)-1); if (!ordered) @@ -8185,8 +8121,7 @@ void btrfs_destroy_inode(struct inode *vfs_inode) "found ordered extent %llu %llu on inode cleanup", ordered->file_offset, ordered->num_bytes); - if (!freespace_inode) - btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); btrfs_remove_ordered_extent(ordered); btrfs_put_ordered_extent(ordered); @@ -8511,14 +8446,14 @@ static int btrfs_rename_exchange(struct inode *old_dir, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - new_name, false, old_idx); + new_name, false, old_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; } ret = btrfs_add_link(trans, BTRFS_I(old_dir), BTRFS_I(new_inode), - old_name, false, new_idx); + old_name, false, new_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8591,6 +8526,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, struct inode *new_inode = d_inode(new_dentry); struct inode *old_inode = d_inode(old_dentry); struct btrfs_rename_ctx rename_ctx; + struct btrfs_dir_index_prealloc *prealloc = NULL; u64 index = 0; int ret; int ret2; @@ -8714,6 +8650,23 @@ static int btrfs_rename(struct mnt_idmap *idmap, if (ret) goto out_fail; + /* + * When not overwriting an existing entry, pre-allocate the delayed dir + * index now so that ENOMEM is returned before any btree modifications. + * For the overwrite case, too many btree changes have already happened + * by the time btrfs_add_link() is called. + */ + if (!new_inode) { + prealloc = btrfs_prealloc_delayed_dir_index(BTRFS_I(new_dir), + new_fname.disk_name.name, + new_fname.disk_name.len); + if (IS_ERR(prealloc)) { + ret = PTR_ERR(prealloc); + prealloc = NULL; + goto out_fail; + } + } + BTRFS_I(old_inode)->dir_index = 0ULL; if (unlikely(old_ino == BTRFS_FIRST_FREE_OBJECTID)) { /* force full log commit if subvolume involved. */ @@ -8809,7 +8762,8 @@ static int btrfs_rename(struct mnt_idmap *idmap, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - &new_fname.disk_name, false, index); + &new_fname.disk_name, false, index, prealloc); + prealloc = NULL; if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8834,6 +8788,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, } } out_fail: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); if (logs_pinned) { btrfs_end_log_trans(root); btrfs_end_log_trans(dest); @@ -9148,14 +9103,13 @@ out_inode: } static struct btrfs_trans_handle *insert_prealloc_file_extent( - struct btrfs_trans_handle *trans_in, struct btrfs_inode *inode, struct btrfs_key *ins, u64 file_offset) { struct btrfs_file_extent_item stack_fi; struct btrfs_replace_extent_info extent_info; - struct btrfs_trans_handle *trans = trans_in; + struct btrfs_trans_handle *trans; struct btrfs_path *path; u64 start = ins->objectid; u64 len = ins->offset; @@ -9176,15 +9130,6 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent( if (ret < 0) return ERR_PTR(ret); - if (trans) { - ret = insert_reserved_file_extent(trans, inode, - file_offset, &stack_fi, - true, qgroup_released); - if (ret) - goto free_qgroup; - return trans; - } - extent_info.disk_offset = start; extent_info.disk_len = len; extent_info.data_offset = 0; @@ -9224,12 +9169,12 @@ free_qgroup: return ERR_PTR(ret); } -static int __btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint, - struct btrfs_trans_handle *trans) +int btrfs_prealloc_file_range(struct inode *inode, int mode, + u64 start, u64 num_bytes, u64 min_size, + loff_t actual_len, u64 *alloc_hint) { struct btrfs_fs_info *fs_info = inode_to_fs_info(inode); + struct btrfs_trans_handle *trans; struct extent_map *em; struct btrfs_root *root = BTRFS_I(inode)->root; struct btrfs_key ins; @@ -9239,11 +9184,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, u64 cur_bytes; u64 last_alloc = (u64)-1; int ret = 0; - bool own_trans = true; u64 end = start + num_bytes - 1; - if (trans) - own_trans = false; while (num_bytes > 0) { cur_bytes = min_t(u64, num_bytes, SZ_256M); cur_bytes = max(cur_bytes, min_size); @@ -9269,8 +9211,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, clear_offset += ins.offset; last_alloc = ins.offset; - trans = insert_prealloc_file_extent(trans, BTRFS_I(inode), - &ins, cur_offset); + trans = insert_prealloc_file_extent(BTRFS_I(inode), &ins, + cur_offset); /* * Now that we inserted the prealloc extent we can finally * decrement the number of reservations in the block group. @@ -9342,8 +9284,7 @@ next: range_start, range_end - range_start); if (ret) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } @@ -9355,15 +9296,11 @@ next: if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } - if (own_trans) { - btrfs_end_transaction(trans); - trans = NULL; - } + btrfs_end_transaction(trans); } if (clear_offset < end) btrfs_free_reserved_data_space(BTRFS_I(inode), NULL, clear_offset, @@ -9371,24 +9308,6 @@ next: return ret; } -int btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, - NULL); -} - -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, trans); -} - /* * NOTE: in case you are adding MAY_EXEC check for directories: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to @@ -9623,7 +9542,6 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, struct completion sync_reads; unsigned long i = 0; struct btrfs_bio *bbio; - int ret; /* * Fast path for synchronous reads which completes in this call, io_uring @@ -9670,10 +9588,10 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, if (uring_ctx) { if (refcount_dec_and_test(&priv->pending_refs)) { - ret = blk_status_to_errno(READ_ONCE(priv->status)); - btrfs_uring_read_extent_endio(uring_ctx, ret); + int err = blk_status_to_errno(READ_ONCE(priv->status)); + + btrfs_uring_read_extent_endio(uring_ctx, err); kfree(priv); - return ret; } return -EIOCBQUEUED; diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index e4b2da31a0d5..52aab510aea0 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -3881,7 +3881,7 @@ static long btrfs_ioctl_quota_rescan_status(struct btrfs_fs_info *fs_info, if (!capable(CAP_SYS_ADMIN)) return -EPERM; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { qsa.flags = 1; qsa.progress = fs_info->qgroup_rescan_progress.objectid; } @@ -4601,7 +4601,7 @@ static void btrfs_uring_read_finished(struct io_tw_req tw_req, io_tw_token_t tw) size_t page_offset; ssize_t ret; - /* The inode lock has already been acquired in btrfs_uring_read_extent. */ + /* The inode lock has already been acquired in btrfs_encoded_read(). */ btrfs_lockdep_inode_acquire(inode, i_rwsem); if (priv->err) { @@ -4667,7 +4667,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, struct iovec *iov, struct io_uring_cmd *cmd) { struct btrfs_inode *inode = BTRFS_I(file_inode(iocb->ki_filp)); - struct extent_io_tree *io_tree = &inode->io_tree; struct page **pages = NULL; struct btrfs_uring_priv *priv = NULL; unsigned long nr_pages; @@ -4723,8 +4722,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, return -EIOCBQUEUED; out_fail: - btrfs_unlock_extent(io_tree, start, lockend, &cached_state); - btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); kfree(priv); for (int i = 0; i < nr_pages; i++) { if (pages[i]) @@ -4752,9 +4749,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4836,8 +4830,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); - if (ret == -EAGAIN) - goto out_acct; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4865,8 +4857,10 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue cached_state, disk_bytenr, disk_io_size, count, data->args.compression, data->iov, cmd); - - goto out_acct; + if (ret == -EIOCBQUEUED) + goto out_acct; + btrfs_unlock_extent(io_tree, start, lockend, &cached_state); + btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); } out_free: @@ -4877,8 +4871,10 @@ out_acct: add_rchar(current, ret); inc_syscr(current); - if (ret != -EIOCBQUEUED && ret != -EAGAIN) + if (ret != -EIOCBQUEUED) { kfree(data); + bc->data = NULL; + } return ret; } @@ -4890,12 +4886,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu struct kiocb kiocb; ssize_t ret; void __user *sqe_addr; - struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4907,6 +4899,11 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } + if (issue_flags & IO_URING_F_NONBLOCK) { + ret = -EAGAIN; + goto out_acct; + } + if (!data) { data = kzalloc_obj(*data, GFP_NOFS); if (!data) { @@ -4914,8 +4911,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } - bc->data = data; - if (issue_flags & IO_URING_F_COMPAT) { #if defined(CONFIG_64BIT) && defined(CONFIG_COMPAT) struct btrfs_ioctl_encoded_io_args_32 args32; @@ -4975,11 +4970,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu } } - if (issue_flags & IO_URING_F_NONBLOCK) { - ret = -EAGAIN; - goto out_acct; - } - pos = data->args.offset; ret = rw_verify_area(WRITE, file, &pos, data->args.len); if (ret < 0) @@ -5005,8 +4995,7 @@ out_acct: add_wchar(current, ret); inc_syscw(current); - if (ret != -EAGAIN) - kfree(data); + kfree(data); return ret; } diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c index b32d4eabe0ab..df74c75d6c29 100644 --- a/fs/btrfs/ordered-data.c +++ b/fs/btrfs/ordered-data.c @@ -417,13 +417,10 @@ static bool can_finish_ordered_extent(struct btrfs_ordered_extent *ordered, static void btrfs_queue_ordered_fn(struct btrfs_ordered_extent *ordered) { - struct btrfs_inode *inode = ordered->inode; - struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct btrfs_workqueue *wq = btrfs_is_free_space_inode(inode) ? - fs_info->endio_freespace_worker : fs_info->endio_write_workers; + struct btrfs_fs_info *fs_info = ordered->inode->root->fs_info; btrfs_init_work(&ordered->work, finish_ordered_fn, NULL); - btrfs_queue_work(wq, &ordered->work); + btrfs_queue_work(fs_info->endio_write_workers, &ordered->work); } void btrfs_finish_ordered_extent(struct btrfs_ordered_extent *ordered, @@ -657,13 +654,6 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) struct btrfs_fs_info *fs_info = root->fs_info; struct rb_node *node; bool pending; - bool freespace_inode; - - /* - * If this is a free space inode the thread has not acquired the ordered - * extents lockdep map. - */ - freespace_inode = btrfs_is_free_space_inode(btrfs_inode); btrfs_lockdep_acquire(fs_info, btrfs_trans_pending_ordered); /* This is paired with alloc_ordered_extent(). */ @@ -738,8 +728,7 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) } spin_unlock(&root->ordered_extent_lock); wake_up(&entry->wait); - if (!freespace_inode) - btrfs_lockdep_release(fs_info, btrfs_ordered_extent); + btrfs_lockdep_release(fs_info, btrfs_ordered_extent); } static void btrfs_run_ordered_extent_work(struct btrfs_work *work) @@ -870,17 +859,10 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, u64 start = entry->file_offset; u64 end = start + entry->num_bytes - 1; struct btrfs_inode *inode = entry->inode; - bool freespace_inode; trace_btrfs_ordered_extent_start(inode, entry); /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - - /* * pages in the range can be dirty, clean or writeback. We * start IO on any dirty ones so the wait doesn't stall waiting * for the flusher thread to find them @@ -899,8 +881,7 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, } } - if (!freespace_inode) - btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); + btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); wait_event(entry->wait, test_bit(BTRFS_ORDERED_COMPLETE, &entry->flags)); } diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f68b696b4bf7..05e35eb126dc 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -34,7 +34,7 @@ enum btrfs_qgroup_mode btrfs_qgroup_mode(const struct btrfs_fs_info *fs_info) { if (!test_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags)) return BTRFS_QGROUP_MODE_DISABLED; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) return BTRFS_QGROUP_MODE_SIMPLE; return BTRFS_QGROUP_MODE_FULL; } @@ -384,14 +384,14 @@ static bool squota_check_parent_usage(struct btrfs_fs_info *fs_info, struct btrf __printf(2, 3) static void qgroup_mark_inconsistent(struct btrfs_fs_info *fs_info, const char *fmt, ...) { - const u64 old_flags = fs_info->qgroup_flags; + const unsigned long old_flags = fs_info->qgroup_flags; if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE) return; - fs_info->qgroup_flags |= (BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT | - BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); - if (!(old_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); + if (!test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &old_flags)) { struct va_format vaf; va_list args; @@ -426,7 +426,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) struct extent_buffer *l; int slot; int ret = 0; - u64 flags = 0; u64 rescan_progress = 0; if (!fs_info->quota_root) @@ -473,8 +472,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) "old qgroup version, quota disabled"); goto out; } - fs_info->qgroup_flags = btrfs_qgroup_status_flags(l, ptr); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (btrfs_qgroup_status_flags(l, ptr) > ULONG_MAX) { + btrfs_err(fs_info, "invalid qgroup status flags, quota disabled"); + goto out; + } + fs_info->qgroup_flags = (unsigned long)btrfs_qgroup_status_flags(l, ptr); + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) qgroup_read_enable_gen(fs_info, l, slot, ptr); else if (btrfs_qgroup_status_generation(l, ptr) != fs_info->generation) qgroup_mark_inconsistent(fs_info, "qgroup generation mismatch"); @@ -609,14 +612,13 @@ next2: } out: btrfs_free_path(path); - fs_info->qgroup_flags |= flags; if (ret >= 0) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) ret = qgroup_rescan_init(fs_info, rescan_progress, 0); } else { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); btrfs_sysfs_del_qgroups(fs_info); } @@ -1101,9 +1103,9 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = BTRFS_QGROUP_STATUS_FLAG_ON; + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); if (simple) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); /* * Set the enable generation to the next transaction, as we cannot @@ -1113,7 +1115,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, */ btrfs_set_qgroup_status_enable_gen(leaf, ptr, trans->transid + 1); } else { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); } btrfs_set_qgroup_status_flags(leaf, ptr, fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAGS_MASK); @@ -1403,8 +1405,14 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + /* + * Clear all on-disk and runtime bits, except RESCAN related ones, that + * are either handled by rescan thread, or the caller who rejects rescan. + */ + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -1554,7 +1562,7 @@ static int quick_update_accounting(struct btrfs_fs_info *fs_info, } out: if (ret) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return ret; } @@ -1875,7 +1883,7 @@ int btrfs_remove_qgroup(struct btrfs_trans_handle *trans, u64 qgroupid) * very frequently. */ if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_FULL && - !(fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + !test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags)) { if (unlikely(qgroup->rfer || qgroup->excl || qgroup->rfer_cmpr || qgroup->excl_cmpr)) { DEBUG_WARN(); @@ -2120,7 +2128,7 @@ int btrfs_qgroup_trace_extent_post(struct btrfs_trans_handle *trans, */ ASSERT(trans != NULL); - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) return 0; ret = btrfs_find_all_roots(&ctx, true); @@ -2740,6 +2748,24 @@ walk_down: return 0; } +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level) +{ + u8 drop_subtree_thres; + + if (btrfs_qgroup_mode(fs_info) != BTRFS_QGROUP_MODE_FULL) + return; + + if (!btrfs_is_fstree(rootid)) + return; + + spin_lock(&fs_info->qgroup_lock); + drop_subtree_thres = fs_info->qgroup_drop_subtree_thres; + spin_unlock(&fs_info->qgroup_lock); + + if (level >= drop_subtree_thres) + qgroup_mark_inconsistent(fs_info, "subtree level reached threshold"); +} + static void qgroup_iterator_nested_add(struct list_head *head, struct btrfs_qgroup *qgroup) { if (!list_empty(&qgroup->nested_iterator)) @@ -2961,7 +2987,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, * we can't just exit here. */ if (!btrfs_qgroup_full_accounting(fs_info) || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) goto out_free; if (new_roots) { @@ -2983,7 +3009,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, num_bytes, nr_old_roots, nr_new_roots); mutex_lock(&fs_info->qgroup_rescan_lock); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { if (fs_info->qgroup_rescan_progress.objectid <= bytenr) { mutex_unlock(&fs_info->qgroup_rescan_lock); ret = 0; @@ -3044,8 +3070,8 @@ int btrfs_qgroup_account_extents(struct btrfs_trans_handle *trans) num_dirty_extents++; trace_btrfs_qgroup_account_extents(fs_info, record, bytenr); - if (!ret && !(fs_info->qgroup_flags & - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)) { + if (!ret && !test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, + &fs_info->qgroup_flags)) { struct btrfs_backref_walk_ctx ctx = { 0 }; ctx.bytenr = bytenr; @@ -3152,9 +3178,9 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans) spin_lock(&fs_info->qgroup_lock); } if (btrfs_qgroup_enabled(fs_info)) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_ON; + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); else - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); spin_unlock(&fs_info->qgroup_lock); ret = update_qgroup_status_item(trans); @@ -3844,7 +3870,7 @@ static bool rescan_should_stop(struct btrfs_fs_info *fs_info) return true; if (!btrfs_qgroup_enabled(fs_info)) return true; - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) return true; return false; } @@ -3894,12 +3920,10 @@ out: btrfs_free_path(path); mutex_lock(&fs_info->qgroup_rescan_lock); - if (ret > 0 && - fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } else if (ret < 0 || stopped) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } + if (ret > 0) + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + else if (ret < 0 || stopped) + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); mutex_unlock(&fs_info->qgroup_rescan_lock); /* @@ -3923,9 +3947,9 @@ out: } mutex_lock(&fs_info->qgroup_rescan_lock); - if (!stopped || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + if (!stopped || test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, + &fs_info->qgroup_flags)) + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); if (trans) { int ret2 = update_qgroup_status_item(trans); @@ -3935,7 +3959,7 @@ out: } } fs_info->qgroup_rescan_running = false; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN; + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); complete_all(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -3946,7 +3970,7 @@ out: if (stopped) { btrfs_info(fs_info, "qgroup scan paused"); - } else if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) { + } else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) { btrfs_info(fs_info, "qgroup scan cancelled"); } else if (ret >= 0) { btrfs_info(fs_info, "qgroup scan completed%s", @@ -3973,13 +3997,11 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, if (!init_flags) { /* we're resuming qgroup rescan at mount time */ - if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_RESCAN)) { + if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup rescan is not queued"); ret = -EINVAL; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -3992,10 +4014,12 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, + &fs_info->qgroup_flags) || + test_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, + &fs_info->qgroup_flags)) { ret = -EINPROGRESS; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -4008,13 +4032,13 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_unlock(&fs_info->qgroup_rescan_lock); return ret; } - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_RESCAN; + set_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); } memset(&fs_info->qgroup_rescan_progress, 0, sizeof(fs_info->qgroup_rescan_progress)); - fs_info->qgroup_flags &= ~(BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_rescan_progress.objectid = progress_objectid; init_completion(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -4065,7 +4089,7 @@ btrfs_qgroup_rescan(struct btrfs_fs_info *fs_info) ret = btrfs_commit_current_transaction(fs_info->fs_root); if (ret) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); return ret; } @@ -4118,7 +4142,7 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info, void btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { mutex_lock(&fs_info->qgroup_rescan_lock); fs_info->qgroup_rescan_running = true; btrfs_queue_work(fs_info->qgroup_rescan_workers, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 80dd2dacd56d..c64b26b09c22 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -121,8 +121,19 @@ struct btrfs_qgroup_swapped_blocks; * To minimize the chance of collision with new persisted status flags, these * count backwards from the MSB. */ -#define BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN (1ULL << 63) -#define BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING (1ULL << 62) +#define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) +#define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) + +/* + * No new rescan allowed when set. + * + * During huge subtree dropping, qgroup will be marked inconsistent, and skip + * all future accounting to avoid long stall. But, an immediate rescan will + * re-enable qgroup and still stall the system. + * + * This bit is to avoid such rescan during the duration of a subvolume dropping. + */ +#define BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN (BITS_PER_LONG - 3) #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) @@ -365,6 +376,7 @@ int btrfs_qgroup_trace_leaf_items(struct btrfs_trans_handle *trans, int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, struct extent_buffer *root_eb, u64 root_gen, int root_level); +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level); int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, struct ulist *old_roots, struct ulist *new_roots); diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index 1ee52a9dcee3..8ec24dbb180f 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -953,7 +953,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) /* * Clear the data bitmap, as the rbio may be cached for later usage. - * do this before before unlock_stripe() so there will be no new bio + * do this before unlock_stripe() so there will be no new bio * for this bio. */ bitmap_clear(&rbio->dbitmap, 0, rbio->stripe_nsectors); @@ -988,7 +988,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) * as possible, and only use stripe_sectors as fallback. * * Return NULL if bio_list_only is set but the specified sector has no - * coresponding bio. + * corresponding bio. */ static phys_addr_t *sector_paddrs_in_rbio(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr, @@ -1450,10 +1450,7 @@ static int rmw_assemble_write_bios(struct btrfs_raid_bio *rbio, /* We should have at least one data sector. */ ASSERT(bitmap_weight(&rbio->dbitmap, rbio->stripe_nsectors)); - /* - * Reset errors, as we may have errors inherited from from degraded - * write. - */ + /* Reset errors, as we may have errors inherited from degraded write. */ bitmap_clear(rbio->error_bitmap, 0, rbio->nr_sectors); /* @@ -1652,12 +1649,7 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, struct bio *bio) { struct btrfs_fs_info *fs_info = rbio->bioc->fs_info; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = rbio->sector_nsteps; int total_sector_nr = get_bio_sector_nr(rbio, bio); - u32 offset = 0; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; /* No data csum for the whole stripe, no need to verify. */ if (!rbio->csum_bitmap || !rbio->csum_buf) @@ -1667,28 +1659,20 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, if (total_sector_nr >= rbio->nr_data * rbio->stripe_nsectors) return; - btrfs_bio_for_each_block_all(paddr, bio, step) { + for (struct bvec_iter iter = init_bvec_iter_for_bio(bio); + iter.bi_size; + bio_advance_iter(bio, &iter, fs_info->sectorsize), total_sector_nr++) { u8 csum_buf[BTRFS_CSUM_SIZE]; u8 *expected_csum; - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - /* Not yet covering the full fs block, continue to the next step. */ - if (!IS_ALIGNED(offset, fs_info->sectorsize)) - continue; - /* No csum for this sector, skip to the next sector. */ - if (!test_bit(total_sector_nr, rbio->csum_bitmap)) { - total_sector_nr++; + if (!test_bit(total_sector_nr, rbio->csum_bitmap)) continue; - } expected_csum = rbio->csum_buf + total_sector_nr * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + btrfs_csum_one_bio_block(fs_info, bio, &iter, csum_buf); if (unlikely(memcmp(csum_buf, expected_csum, fs_info->csum_size) != 0)) set_bit(total_sector_nr, rbio->error_bitmap); - total_sector_nr++; } } @@ -1879,6 +1863,27 @@ void raid56_parity_write(struct bio *bio, struct btrfs_io_context *bioc) start_async_work(rbio, rmw_rbio_work); } +static void calculate_block_csum_paddrs(struct btrfs_fs_info *fs_info, + const phys_addr_t paddrs[], u8 *dest) +{ + const u32 blocksize = fs_info->sectorsize; + const u32 step = min(blocksize, PAGE_SIZE); + const u32 nr_steps = blocksize / step; + struct btrfs_csum_ctx csum; + + btrfs_csum_init(&csum, fs_info->csum_type); + for (int i = 0; i < nr_steps; i++) { + const phys_addr_t paddr = paddrs[i]; + void *kaddr; + + ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); + kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); + btrfs_csum_update(&csum, kaddr, step); + kunmap_local(kaddr); + } + btrfs_csum_final(&csum, dest); +} + static int verify_one_sector(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr) { @@ -1906,7 +1911,7 @@ static int verify_one_sector(struct btrfs_raid_bio *rbio, csum_expected = rbio->csum_buf + (stripe_nr * rbio->stripe_nsectors + sector_nr) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + calculate_block_csum_paddrs(fs_info, paddrs, csum_buf); if (unlikely(memcmp(csum_buf, csum_expected, fs_info->csum_size) != 0)) return -EIO; return 0; @@ -2624,7 +2629,7 @@ static int alloc_rbio_essential_pages(struct btrfs_raid_bio *rbio) return 0; } -/* Return true if the content of the step matches the caclulated one. */ +/* Return true if the content of the step matches the calculated one. */ static bool verify_one_parity_step(struct btrfs_raid_bio *rbio, void *pointers[], unsigned int sector_nr, unsigned int step_nr) diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c index da54db75e7a9..630a7ad8f8e1 100644 --- a/fs/btrfs/relocation.c +++ b/fs/btrfs/relocation.c @@ -3357,7 +3357,7 @@ truncate: goto out; } - ret = btrfs_truncate_free_space_cache(trans, block_group, inode); + ret = btrfs_truncate_free_space_cache(trans, inode); btrfs_end_transaction(trans); btrfs_btree_balance_dirty(fs_info); diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c index 5c59b9abedcd..c523bf950c89 100644 --- a/fs/btrfs/send.c +++ b/fs/btrfs/send.c @@ -7023,7 +7023,7 @@ static int changed_extent(struct send_ctx *sctx, * get modified or replaced with a new one). Note that deduplication * updates the inode item, but it only changes the iversion (sequence * field in the inode item) of the inode, so if a file is deduplicated - * the same amount of times in both the parent and send snapshots, its + * the same number of times in both the parent and send snapshots, its * iversion becomes the same in both snapshots, whence the inode item is * the same on both snapshots. */ diff --git a/fs/btrfs/space-info.c b/fs/btrfs/space-info.c index 39a28e1bec8a..01018152c054 100644 --- a/fs/btrfs/space-info.c +++ b/fs/btrfs/space-info.c @@ -1704,7 +1704,6 @@ static int handle_reserve_ticket(struct btrfs_space_info *space_info, evict_flush_states, ARRAY_SIZE(evict_flush_states)); break; - case BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE: case BTRFS_RESERVE_FLUSH_ZONED_RELOCATION: priority_reclaim_data_space(space_info, ticket); break; @@ -1968,7 +1967,6 @@ int btrfs_reserve_data_bytes(struct btrfs_space_info *space_info, u64 bytes, int ret; ASSERT(flush == BTRFS_RESERVE_FLUSH_DATA || - flush == BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE || flush == BTRFS_RESERVE_FLUSH_ZONED_RELOCATION || flush == BTRFS_RESERVE_NO_FLUSH, "flush=%d", flush); ASSERT(!current->journal_info || flush != BTRFS_RESERVE_FLUSH_DATA, diff --git a/fs/btrfs/space-info.h b/fs/btrfs/space-info.h index aa836e8a9d4a..d0130c8ba3dd 100644 --- a/fs/btrfs/space-info.h +++ b/fs/btrfs/space-info.h @@ -66,7 +66,6 @@ enum btrfs_reserve_flush_enum { * Can be interrupted by a fatal signal. */ BTRFS_RESERVE_FLUSH_DATA, - BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE, BTRFS_RESERVE_FLUSH_ALL, /* @@ -82,9 +81,6 @@ enum btrfs_reserve_flush_enum { * priority flushing for this, because otherwise we can deadlock on * waiting for a ticket, that cannot be granted, because we cannot do * any allocations. - * - * Apart from being specific to zoned relocation, it is equal to - * BTRFS_FLUSH_FREE_SPACE_INODE. */ BTRFS_RESERVE_FLUSH_ZONED_RELOCATION, diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index ddb620ac241b..14ed0ed823a3 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -515,7 +515,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_version: @@ -524,11 +523,9 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_v2: - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_set_opt(ctx->mount_opt, FREE_SPACE_TREE); break; default: @@ -705,13 +702,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, if (btrfs_check_mountopts_zoned(info, mount_opt)) ret = false; - if (!test_bit(BTRFS_FS_STATE_REMOUNTING, &info->fs_state)) { - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_warn(info, -"space cache v1 is being deprecated and will be removed in a future release, please use -o space_cache=v2"); - } - } - return ret; } @@ -729,14 +719,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, */ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) { - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_info(fs_info, - "forcing free space tree for sector size %u with page size %lu", - fs_info->sectorsize, PAGE_SIZE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - } - /* * At this point our mount options are populated, so we only mess with * these settings if we don't have any settings already. @@ -751,20 +733,17 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) return; } - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - if (btrfs_test_opt(fs_info, NOSPACECACHE)) return; /* * At this point we don't have explicit options set by the user, set - * them ourselves based on the state of the file system. + * them ourselves based on the state of the file system. An existing + * v1 space cache is no longer used and gets cleaned up once the + * filesystem is mounted read-write. */ if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - else if (btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); } static void set_device_specific_options(struct btrfs_fs_info *fs_info) @@ -1107,9 +1086,7 @@ static int btrfs_show_options(struct seq_file *seq, struct dentry *dentry) seq_puts(seq, ",discard=async"); if (!(info->sb->s_flags & SB_POSIXACL)) seq_puts(seq, ",noacl"); - if (btrfs_free_space_cache_v1_active(info)) - seq_puts(seq, ",space_cache"); - else if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) + if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) seq_puts(seq, ",space_cache=v2"); else seq_puts(seq, ",nospace_cache"); @@ -1243,7 +1220,6 @@ static void btrfs_resize_thread_pool(struct btrfs_fs_info *fs_info, workqueue_set_max_active(fs_info->endio_workers, new_pool_size); workqueue_set_max_active(fs_info->endio_meta_workers, new_pool_size); btrfs_workqueue_set_max(fs_info->endio_write_workers, new_pool_size); - btrfs_workqueue_set_max(fs_info->endio_freespace_worker, new_pool_size); btrfs_workqueue_set_max(fs_info->delayed_workers, new_pool_size); } @@ -1264,8 +1240,6 @@ static inline void btrfs_remount_begin(struct btrfs_fs_info *fs_info, static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, unsigned long long old_opts) { - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); - /* * We need to cleanup all defraggable inodes if the autodefragment is * close or the filesystem is read only. @@ -1282,10 +1256,6 @@ static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, else if (btrfs_raw_test_opt(old_opts, DISCARD_ASYNC) && !btrfs_test_opt(fs_info, DISCARD_ASYNC)) btrfs_discard_cleanup(fs_info); - - /* If we toggled space cache */ - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); } static int btrfs_remount_rw(struct btrfs_fs_info *fs_info) @@ -1448,7 +1418,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_set(info, old, DISCARD_SYNC, "turning on sync discard"); btrfs_info_if_set(info, old, DISCARD_ASYNC, "turning on async discard"); btrfs_info_if_set(info, old, FREE_SPACE_TREE, "enabling free space tree"); - btrfs_info_if_set(info, old, SPACE_CACHE, "enabling disk space caching"); btrfs_info_if_set(info, old, CLEAR_CACHE, "force clearing of disk cache"); btrfs_info_if_set(info, old, AUTO_DEFRAG, "enabling auto defrag"); btrfs_info_if_set(info, old, FRAGMENT_DATA, "fragmenting data"); @@ -1466,7 +1435,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_unset(info, old, SSD_SPREAD, "not using spread ssd allocation scheme"); btrfs_info_if_unset(info, old, NOBARRIER, "turning on barriers"); btrfs_info_if_unset(info, old, NOTREELOG, "enabling tree log"); - btrfs_info_if_unset(info, old, SPACE_CACHE, "disabling disk space caching"); btrfs_info_if_unset(info, old, FREE_SPACE_TREE, "disabling free space tree"); btrfs_info_if_unset(info, old, AUTO_DEFRAG, "disabling auto defrag"); btrfs_info_if_unset(info, old, COMPRESS, "use no compression"); @@ -1531,14 +1499,8 @@ static int btrfs_reconfigure(struct fs_context *fc) btrfs_warn(fs_info, "remount supports changing free space tree only from RO to RW"); /* Make sure free space cache options match the state on disk. */ - if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) { + if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - } - if (btrfs_free_space_cache_v1_active(fs_info)) { - btrfs_clear_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); - } } ret = 0; diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 39cb01ee441a..c5bb1c7eac6a 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -83,8 +83,7 @@ struct raid_kobject { #define BTRFS_FEAT_ATTR(_name, _feature_set, _feature_prefix, _feature_bit) \ static struct btrfs_feature_attr btrfs_attr_features_##_name = { \ .kobj_attr = __INIT_KOBJ_ATTR(_name, S_IRUGO, \ - btrfs_feature_attr_show, \ - btrfs_feature_attr_store), \ + btrfs_feature_attr_show, NULL), \ .feature_set = _feature_set, \ .feature_bit = _feature_prefix ##_## _feature_bit, \ } @@ -130,130 +129,20 @@ static u64 get_features(struct btrfs_fs_info *fs_info, return btrfs_super_incompat_flags(disk_super); } -static void set_features(struct btrfs_fs_info *fs_info, - enum btrfs_feature_set set, u64 features) -{ - struct btrfs_super_block *disk_super = fs_info->super_copy; - if (set == FEAT_COMPAT) - btrfs_set_super_compat_flags(disk_super, features); - else if (set == FEAT_COMPAT_RO) - btrfs_set_super_compat_ro_flags(disk_super, features); - else - btrfs_set_super_incompat_flags(disk_super, features); -} - -static int can_modify_feature(struct btrfs_feature_attr *fa) -{ - int val = 0; - u64 set, clear; - switch (fa->feature_set) { - case FEAT_COMPAT: - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - break; - case FEAT_COMPAT_RO: - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - break; - case FEAT_INCOMPAT: - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; - break; - default: - btrfs_warn(NULL, "sysfs: unknown feature set %d", fa->feature_set); - return 0; - } - - if (set & fa->feature_bit) - val |= 1; - if (clear & fa->feature_bit) - val |= 2; - - return val; -} - static ssize_t btrfs_feature_attr_show(struct kobject *kobj, struct kobj_attribute *a, char *buf) { int val = 0; struct btrfs_fs_info *fs_info = to_fs_info(kobj); struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); + if (fs_info) { u64 features = get_features(fs_info, fa->feature_set); if (features & fa->feature_bit) val = 1; - } else - val = can_modify_feature(fa); - - return sysfs_emit(buf, "%d\n", val); -} - -static ssize_t btrfs_feature_attr_store(struct kobject *kobj, - struct kobj_attribute *a, - const char *buf, size_t count) -{ - struct btrfs_fs_info *fs_info; - struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); - u64 features, set, clear; - unsigned long val; - int ret; - - fs_info = to_fs_info(kobj); - if (!fs_info) - return -EPERM; - - if (sb_rdonly(fs_info->sb)) - return -EROFS; - - ret = kstrtoul(skip_spaces(buf), 0, &val); - if (ret) - return ret; - - if (fa->feature_set == FEAT_COMPAT) { - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - } else if (fa->feature_set == FEAT_COMPAT_RO) { - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - } else { - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; - } - - features = get_features(fs_info, fa->feature_set); - - /* Nothing to do */ - if ((val && (features & fa->feature_bit)) || - (!val && !(features & fa->feature_bit))) - return count; - - if ((val && !(set & fa->feature_bit)) || - (!val && !(clear & fa->feature_bit))) { - btrfs_info(fs_info, - "%sabling feature %s on mounted fs is not supported.", - val ? "En" : "Dis", fa->kobj_attr.attr.name); - return -EPERM; } - btrfs_info(fs_info, "%s %s feature flag", - val ? "Setting" : "Clearing", fa->kobj_attr.attr.name); - - spin_lock(&fs_info->super_lock); - features = get_features(fs_info, fa->feature_set); - if (val) - features |= fa->feature_bit; - else - features &= ~fa->feature_bit; - set_features(fs_info, fa->feature_set, features); - spin_unlock(&fs_info->super_lock); - - /* - * We don't want to do full transaction commit from inside sysfs - */ - set_bit(BTRFS_FS_NEED_TRANS_COMMIT, &fs_info->flags); - wake_up_process(fs_info->transaction_kthread); - - return count; + return sysfs_emit(buf, "%d\n", val); } static umode_t btrfs_feature_visible(struct kobject *kobj, @@ -269,9 +158,7 @@ static umode_t btrfs_feature_visible(struct kobject *kobj, fa = attr_to_btrfs_feature_attr(attr); features = get_features(fs_info, fa->feature_set); - if (can_modify_feature(fa)) - mode |= S_IWUSR; - else if (!(features & fa->feature_bit)) + if (!(features & fa->feature_bit)) mode = 0; } @@ -2359,9 +2246,7 @@ static ssize_t qgroup_enabled_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool enabled; - spin_lock(&fs_info->qgroup_lock); - enabled = fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON; - spin_unlock(&fs_info->qgroup_lock); + enabled = test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", enabled); } @@ -2401,9 +2286,7 @@ static ssize_t qgroup_inconsistent_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool inconsistent; - spin_lock(&fs_info->qgroup_lock); - inconsistent = (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT); - spin_unlock(&fs_info->qgroup_lock); + inconsistent = test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", inconsistent); } diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index 23459cd4e503..cd045778400d 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -18,8 +18,8 @@ #define PROCESS_RELEASE (1U << 1) #define PROCESS_TEST_LOCKED (1U << 2) -static noinline int process_page_range(struct inode *inode, u64 start, u64 end, - unsigned long flags) +static noinline int process_folio_range(struct inode *inode, u64 start, u64 end, + unsigned long flags) { int ret; struct folio_batch fbatch; @@ -112,8 +112,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) struct btrfs_root *root = NULL; struct inode *inode = NULL; struct extent_io_tree *tmp; - struct page *page; - struct page *locked_page = NULL; + struct folio *folio; + struct folio *locked_folio = NULL; /* In this test we need at least 2 file extents at its maximum size */ u64 max_bytes = BTRFS_MAX_EXTENT_SIZE; u64 total_dirty = 2 * max_bytes; @@ -152,23 +152,27 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_extent_io_tree_init(NULL, tmp, IO_TREE_SELFTEST); /* - * First go through and create and mark all of our pages dirty, we pin - * everything to make sure our pages don't get evicted and screw up our + * First go through and create and mark all of our folios dirty, we pin + * everything to make sure our folios don't get evicted and screw up our * test. */ for (pgoff_t index = 0; index < (total_dirty >> PAGE_SHIFT); index++) { - page = find_or_create_page(inode->i_mapping, index, GFP_KERNEL); - if (!page) { - test_err("failed to allocate test page"); - ret = -ENOMEM; + folio = __filemap_get_folio(inode->i_mapping, index, + FGP_LOCK | FGP_ACCESSED | FGP_CREAT, + GFP_KERNEL); + if (IS_ERR(folio)) { + test_err("failed to allocate test folio"); + ret = PTR_ERR(folio); goto out; } - SetPageDirty(page); + /* The ranges below assume page sized folios. */ + ASSERT(folio_order(folio) == 0); + folio_set_dirty(folio); if (index) { - unlock_page(page); + folio_unlock(folio); } else { - get_page(page); - locked_page = page; + folio_get(folio); + locked_folio = folio; } } @@ -179,8 +183,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, 0, sectorsize - 1, EXTENT_DELALLOC, NULL); start = 0; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("should have found at least one delalloc"); goto out_bits; @@ -191,8 +194,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - unlock_page(locked_page); - put_page(locked_page); + folio_unlock(locked_folio); + folio_put(locked_folio); /* * Test this scenario @@ -201,17 +204,17 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = SZ_64M; - locked_page = find_lock_page(inode->i_mapping, - test_start >> PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); btrfs_set_extent_bit(tmp, sectorsize, max_bytes - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("couldn't find delalloc in our range"); goto out_bits; @@ -221,14 +224,14 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, max_bytes - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("there were unlocked pages in the range"); + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + test_err("there were unlocked folios in the range"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - /* locked_page was unlocked above */ - put_page(locked_page); + /* locked_folio was unlocked above */ + folio_put(locked_folio); /* * Test this scenario @@ -236,16 +239,16 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = max_bytes + sectorsize; - locked_page = find_lock_page(inode->i_mapping, test_start >> - PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (found) { test_err("found range when we shouldn't have"); goto out_bits; @@ -265,8 +268,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, max_bytes, total_dirty - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -276,38 +278,37 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, total_dirty - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + test_err("folios in range were not all locked"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); /* - * Now to test where we run into a page that is no longer dirty in the + * Now to test where we run into a folio that is no longer dirty in the * range we want to find. */ - page = find_get_page(inode->i_mapping, - (max_bytes + SZ_1M) >> PAGE_SHIFT); - if (!page) { - test_err("couldn't find our page"); + folio = filemap_get_folio(inode->i_mapping, (max_bytes + SZ_1M) >> PAGE_SHIFT); + if (IS_ERR(folio)) { + test_err("couldn't find our folio"); goto out_bits; } - ClearPageDirty(page); - put_page(page); + ASSERT(folio_order(folio) == 0); + folio_clear_dirty(folio); + folio_put(folio); /* We unlocked it in the previous test */ - lock_page(locked_page); + folio_lock(locked_folio); start = test_start; end = start + PAGE_SIZE - 1; /* - * Currently if we fail to find dirty pages in the delalloc range we + * Currently if we fail to find dirty folios in the delalloc range we * will adjust max_bytes down to PAGE_SIZE and then re-search. If * this changes at any point in the future we will need to fix this * tests expected behavior. */ - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -317,9 +318,9 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, test_start + PAGE_SIZE - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, PROCESS_TEST_LOCKED | - PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | + PROCESS_UNLOCK)) { + test_err("folios in range were not all locked"); goto out_bits; } ret = 0; @@ -328,10 +329,10 @@ out_bits: dump_extent_io_tree(tmp); btrfs_clear_extent_bit(tmp, 0, total_dirty - 1, (unsigned)-1, NULL); out: - if (locked_page) - put_page(locked_page); - process_page_range(inode, 0, total_dirty - 1, - PROCESS_UNLOCK | PROCESS_RELEASE); + if (locked_folio) + folio_put(locked_folio); + process_folio_range(inode, 0, total_dirty - 1, + PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); out_root_info: btrfs_free_dummy_root(root); @@ -671,8 +672,9 @@ static void dump_eb_and_memory_contents(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < eb->len; i++) { - struct page *page = folio_page(eb->folios[i >> PAGE_SHIFT], 0); - void *addr = page_address(page) + offset_in_page(i); + const unsigned long idx = get_eb_folio_index(eb, i); + void *addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, i); if (memcmp(addr, memory + i, 1) != 0) { test_err("%s failed", test_name); @@ -687,9 +689,12 @@ static int verify_eb_and_memory(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < (eb->len >> PAGE_SHIFT); i++) { - void *eb_addr = folio_address(eb->folios[i]); + const unsigned long offset = i << PAGE_SHIFT; + const unsigned long idx = get_eb_folio_index(eb, offset); + void *eb_addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, offset); - if (memcmp(memory + (i << PAGE_SHIFT), eb_addr, PAGE_SIZE) != 0) { + if (memcmp(memory + offset, eb_addr, PAGE_SIZE) != 0) { dump_eb_and_memory_contents(eb, memory, test_name); return -EUCLEAN; } diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 6802b94ed76f..84f012bfcffc 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -125,17 +125,14 @@ static const unsigned int btrfs_blocked_trans_types[TRANS_STATE_MAX] = { [TRANS_STATE_UNBLOCKED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), [TRANS_STATE_SUPER_COMMITTED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), [TRANS_STATE_COMPLETED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), }; @@ -310,12 +307,6 @@ loop: if (type == TRANS_ATTACH || type == TRANS_JOIN_NOSTART) return -ENOENT; - /* - * JOIN_NOLOCK only happens during the transaction commit, so - * it is impossible that ->running_transaction is NULL - */ - BUG_ON(type == TRANS_JOIN_NOLOCK); - cur_trans = kmalloc_obj(*cur_trans, GFP_NOFS); if (!cur_trans) return -ENOMEM; @@ -379,9 +370,8 @@ loop: INIT_LIST_HEAD(&cur_trans->dev_update_list); INIT_LIST_HEAD(&cur_trans->switch_commits); INIT_LIST_HEAD(&cur_trans->dirty_bgs); - INIT_LIST_HEAD(&cur_trans->io_bgs); INIT_LIST_HEAD(&cur_trans->dropped_roots); - mutex_init(&cur_trans->cache_write_mutex); + mutex_init(&cur_trans->dirty_bgs_update_mutex); spin_lock_init(&cur_trans->dirty_bgs_lock); INIT_LIST_HEAD(&cur_trans->deleted_bgs); spin_lock_init(&cur_trans->dropped_roots_lock); @@ -710,14 +700,8 @@ again: } /* - * If we are JOIN_NOLOCK we're already committing a transaction and - * waiting on this guy, so we don't need to do the sb_start_intwrite - * because we're already holding a ref. We need this because we could - * have raced in and did an fsync() on a file which can kick a commit - * and then we deadlock with somebody doing a freeze. - * * If we are ATTACH, it means we just want to catch the current - * transaction and commit it, so we needn't do sb_start_intwrite(). + * transaction and commit it, so we needn't do sb_start_intwrite(). */ if (type & __TRANS_FREEZABLE) sb_start_intwrite(fs_info->sb); @@ -855,12 +839,6 @@ struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root) true); } -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root) -{ - return start_transaction(root, 0, TRANS_JOIN_NOLOCK, - BTRFS_RESERVE_NO_FLUSH, true); -} - /* * Similar to regular join but it never starts a transaction when none is * running or when there's a running one at a state >= TRANS_STATE_UNBLOCKED. @@ -1363,7 +1341,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) { struct btrfs_fs_info *fs_info = trans->fs_info; struct list_head *dirty_bgs = &trans->transaction->dirty_bgs; - struct list_head *io_bgs = &trans->transaction->io_bgs; struct extent_buffer *eb; int ret; @@ -1393,10 +1370,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) if (ret) return ret; - ret = btrfs_setup_space_cache(trans); - if (ret) - return ret; - again: while (!list_empty(&fs_info->dirty_cowonly_roots)) { struct btrfs_root *root; @@ -1417,7 +1390,7 @@ again: if (ret) return ret; - while (!list_empty(dirty_bgs) || !list_empty(io_bgs)) { + while (!list_empty(dirty_bgs)) { ret = btrfs_write_dirty_block_groups(trans); if (ret) return ret; @@ -1890,8 +1863,7 @@ static noinline int create_pending_snapshot(struct btrfs_trans_handle *trans, goto fail; ret = btrfs_insert_dir_item(trans, &fname.disk_name, - parent_inode, &key, BTRFS_FT_DIR, - index); + parent_inode, &key, BTRFS_FT_DIR, index, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto fail; @@ -1990,9 +1962,7 @@ static void update_super_roots(struct btrfs_fs_info *fs_info) super->root = root_item->bytenr; super->generation = root_item->generation; super->root_level = root_item->level; - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - super->cache_generation = root_item->generation; - else if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) + if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) super->cache_generation = 0; if (test_bit(BTRFS_FS_UPDATE_UUID_TREE_GEN, &fs_info->flags)) super->uuid_tree_generation = root_item->generation; @@ -2274,18 +2244,16 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (!test_bit(BTRFS_TRANS_DIRTY_BG_RUN, &cur_trans->flags)) { bool run_it = false; - /* this mutex is also taken before trying to set - * block groups readonly. We need to make sure - * that nobody has set a block group readonly - * after a extents from that block group have been - * allocated for cache files. btrfs_set_block_group_ro - * will wait for the transaction to commit if it - * finds BTRFS_TRANS_DIRTY_BG_RUN set. + /* + * This mutex is also taken before trying to set block groups + * readonly. btrfs_inc_block_group_ro() will wait for the + * transaction to commit if it finds BTRFS_TRANS_DIRTY_BG_RUN + * set. * * The BTRFS_TRANS_DIRTY_BG_RUN flag is also used to make sure - * only one process starts all the block group IO. It wouldn't - * hurt to have more than one go through, but there's no - * real advantage to it either. + * only one process starts all the block group item updates. It + * wouldn't hurt to have more than one go through, but there's + * no real advantage to it either. */ mutex_lock(&fs_info->ro_block_group_mutex); if (!test_and_set_bit(BTRFS_TRANS_DIRTY_BG_RUN, @@ -2519,10 +2487,7 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (unlikely(ret)) goto unlock_reloc; - /* - * The tasks which save the space cache and inode cache may also - * update ->aborted, check it. - */ + /* Other tasks may also have updated ->aborted, check it. */ if (TRANS_ABORTED(cur_trans)) { ret = cur_trans->aborted; goto unlock_reloc; @@ -2543,7 +2508,6 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) switch_commit_roots(trans); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); update_super_roots(fs_info); btrfs_set_super_log_root(fs_info->super_copy, 0); diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 3a57f227b5ed..68c724e70809 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -47,7 +47,6 @@ enum btrfs_trans_state { #define BTRFS_TRANS_HAVE_FREE_BGS 0 #define BTRFS_TRANS_DIRTY_BG_RUN 1 -#define BTRFS_TRANS_CACHE_ENOSPC 2 struct btrfs_transaction { u64 transid; @@ -78,32 +77,15 @@ struct btrfs_transaction { struct list_head dev_update_list; struct list_head switch_commits; struct list_head dirty_bgs; - - /* - * There is no explicit lock which protects io_bgs, rather its - * consistency is implied by the fact that all the sites which modify - * it do so under some form of transaction critical section, namely: - * - * - btrfs_start_dirty_block_groups - This function can only ever be - * run by one of the transaction committers. Refer to - * BTRFS_TRANS_DIRTY_BG_RUN usage in btrfs_commit_transaction - * - * - btrfs_write_dirty_blockgroups - this is called by - * commit_cowonly_roots from transaction critical section - * (TRANS_STATE_COMMIT_DOING) - * - * - btrfs_cleanup_dirty_bgs - called on transaction abort - */ - struct list_head io_bgs; struct list_head dropped_roots; struct extent_io_tree pinned_extents; /* - * we need to make sure block group deletion doesn't race with - * free space cache writeout. This mutex keeps them from stomping - * on each other + * We need to make sure block group deletion doesn't race with the + * dirty block group item updates done outside the commit critical + * section. This mutex keeps them from stomping on each other. */ - struct mutex cache_write_mutex; + struct mutex dirty_bgs_update_mutex; spinlock_t dirty_bgs_lock; /* Protected by spin lock fs_info->unused_bgs_lock. */ struct list_head deleted_bgs; @@ -124,7 +106,6 @@ enum { ENUM_BIT(__TRANS_START), ENUM_BIT(__TRANS_ATTACH), ENUM_BIT(__TRANS_JOIN), - ENUM_BIT(__TRANS_JOIN_NOLOCK), ENUM_BIT(__TRANS_DUMMY), ENUM_BIT(__TRANS_JOIN_NOSTART), }; @@ -132,7 +113,6 @@ enum { #define TRANS_START (__TRANS_START | __TRANS_FREEZABLE) #define TRANS_ATTACH (__TRANS_ATTACH) #define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE) -#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) #define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) #define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) @@ -288,7 +268,7 @@ do { \ * Call btrfs_abort_transaction() as early as possible when an error condition * is detected, that way the exact stack trace is reported for some errors. * - * Error number must be negative as it encodes wheather it's the first abort. + * Error number must be negative as it encodes whether it's the first abort. */ #define btrfs_abort_transaction(trans, error) \ do { \ @@ -312,7 +292,6 @@ struct btrfs_trans_handle *btrfs_start_transaction_fallback_global_rsv( struct btrfs_root *root, unsigned int num_items); struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root); -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_join_transaction_nostart(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction_barrier( diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 4b1e47173c63..9cd79d97b9b5 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -108,13 +108,14 @@ static void file_extent_err(const struct extent_buffer *eb, int slot, */ #define CHECK_FE_ALIGNED(leaf, slot, fi, name, alignment) \ ({ \ - if (unlikely(!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)))) \ + const u64 val = btrfs_file_extent_##name((leaf), (fi)); \ + const bool not_aligned = !IS_ALIGNED(val, (alignment)); \ + \ + if (unlikely(not_aligned)) \ file_extent_err((leaf), (slot), \ "invalid %s for file extent, have %llu, should be aligned to %u", \ - (#name), btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)); \ - (!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), (alignment))); \ + (#name), val, (alignment)); \ + not_aligned; \ }) static u64 file_extent_end(struct extent_buffer *leaf, @@ -163,6 +164,12 @@ static void dir_item_err(const struct extent_buffer *eb, int slot, va_end(args); } +/* Record info for the last hit inode. */ +struct saved_inode_info { + u64 ino; + u32 mode; +}; + /* * This functions checks prev_key->objectid, to ensure current key and prev_key * share the same objectid as inode number. @@ -204,15 +211,41 @@ static bool check_prev_ino(struct extent_buffer *leaf, prev_key->objectid, key->objectid); return false; } + +static bool can_have_extent_data(struct extent_buffer *leaf, + struct btrfs_key *key, int slot, u8 fi_type, + const struct saved_inode_info *inode_info) +{ + /* No inode item in this leaf. */ + if (inode_info->ino != key->objectid) + return true; + if (S_ISREG(inode_info->mode)) + return true; + if (S_ISLNK(inode_info->mode)) { + /* For a symlink, the file extent item should always be inlined. */ + if (unlikely(fi_type != BTRFS_FILE_EXTENT_INLINE)) + return false; + return true; + } + + /* + * The rest are special files, e.g. block/FIFO files, which cannot + * have any file extent. + */ + return false; +} + static int check_extent_data_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + const struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_file_extent_item *fi; u32 sectorsize = fs_info->sectorsize; u32 item_size = btrfs_item_size(leaf, slot); u64 extent_end; + u8 fi_type; if (unlikely(!IS_ALIGNED(key->offset, sectorsize))) { file_extent_err(leaf, slot, @@ -243,12 +276,18 @@ static int check_extent_data_item(struct extent_buffer *leaf, SZ_4K); return -EUCLEAN; } - if (unlikely(btrfs_file_extent_type(leaf, fi) >= - BTRFS_NR_FILE_EXTENT_TYPES)) { + fi_type = btrfs_file_extent_type(leaf, fi); + if (unlikely(fi_type >= BTRFS_NR_FILE_EXTENT_TYPES)) { file_extent_err(leaf, slot, "invalid type for file extent, have %u expect range [0, %u]", - btrfs_file_extent_type(leaf, fi), - BTRFS_NR_FILE_EXTENT_TYPES - 1); + fi_type, BTRFS_NR_FILE_EXTENT_TYPES - 1); + return -EUCLEAN; + } + + if (unlikely(!can_have_extent_data(leaf, key, slot, fi_type, inode_info))) { + file_extent_err(leaf, slot, + "unexpected file extent item type %u for inode mode 0%o", + fi_type, inode_info->mode); return -EUCLEAN; } @@ -270,7 +309,8 @@ static int check_extent_data_item(struct extent_buffer *leaf, btrfs_file_extent_encryption(leaf, fi)); return -EUCLEAN; } - if (btrfs_file_extent_type(leaf, fi) == BTRFS_FILE_EXTENT_INLINE) { + + if (fi_type == BTRFS_FILE_EXTENT_INLINE) { /* Inline extent must have 0 as key offset */ if (unlikely(key->offset)) { file_extent_err(leaf, slot, @@ -1081,9 +1121,9 @@ int btrfs_check_chunk_valid(const struct btrfs_fs_info *fs_info, return -EUCLEAN; } - if (!remapped && - !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, - num_stripes, sub_stripes)) { + if (unlikely(!remapped && + !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, + num_stripes, sub_stripes))) { chunk_err(fs_info, leaf, chunk, logical, "invalid num_stripes:sub_stripes %u:%u for profile %llu", num_stripes, sub_stripes, @@ -1206,7 +1246,8 @@ static int check_dev_item(struct extent_buffer *leaf, } static int check_inode_item(struct extent_buffer *leaf, - struct btrfs_key *key, int slot) + struct btrfs_key *key, int slot, + struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_inode_item *iitem; @@ -1291,6 +1332,8 @@ static int check_inode_item(struct extent_buffer *leaf, ro_flags); return -EUCLEAN; } + inode_info->ino = key->objectid; + inode_info->mode = mode; return 0; } @@ -2348,14 +2391,15 @@ static int check_free_space_bitmap(struct extent_buffer *leaf, static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + struct saved_inode_info *inode_info) { int ret = 0; struct btrfs_chunk *chunk; switch (key->type) { case BTRFS_EXTENT_DATA_KEY: - ret = check_extent_data_item(leaf, key, slot, prev_key); + ret = check_extent_data_item(leaf, key, slot, prev_key, inode_info); break; case BTRFS_EXTENT_CSUM_KEY: ret = check_csum_item(leaf, key, slot, prev_key); @@ -2385,7 +2429,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, ret = check_dev_extent_item(leaf, key, slot, prev_key); break; case BTRFS_INODE_ITEM_KEY: - ret = check_inode_item(leaf, key, slot); + ret = check_inode_item(leaf, key, slot, inode_info); break; case BTRFS_ROOT_ITEM_KEY: ret = check_root_item(leaf, key, slot); @@ -2433,6 +2477,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) { struct btrfs_fs_info *fs_info = leaf->fs_info; + struct saved_inode_info inode_info = { 0 }; /* No valid key type is 0, so all key should be larger than this key */ struct btrfs_key prev_key = {0, 0, 0}; struct btrfs_key key; @@ -2568,7 +2613,7 @@ enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) } /* Check if the item size and content meet other criteria. */ - ret = check_leaf_item(leaf, &key, slot, &prev_key); + ret = check_leaf_item(leaf, &key, slot, &prev_key, &inode_info); if (unlikely(ret != BTRFS_TREE_BLOCK_CLEAN)) return ret; diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index a00094604e54..1d9dfb63fc5a 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -503,7 +503,7 @@ static int overwrite_item(struct walk_control *wc) btrfs_release_path(wc->subvol_path); return 0; } - src_copy = kmalloc(item_size, GFP_NOFS); + src_copy = kvmalloc(item_size, GFP_NOFS); if (!src_copy) { btrfs_abort_log_replay(wc, -ENOMEM, "failed to allocate memory for log leaf item"); @@ -514,7 +514,7 @@ static int overwrite_item(struct walk_control *wc) dst_ptr = btrfs_item_ptr_offset(dst_eb, dst_slot); ret = memcmp_extent_buffer(dst_eb, src_copy, dst_ptr, item_size); - kfree(src_copy); + kvfree(src_copy); /* * they have the same contents, just return, this saves * us from cowing blocks in the destination tree and doing @@ -1683,7 +1683,7 @@ static noinline int add_inode_ref(struct walk_control *wc) } /* insert our name */ - ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index); + ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index, NULL); if (ret) { btrfs_abort_log_replay(wc, ret, "failed to add link for inode %llu in dir %llu ref_index %llu name %.*s root %llu", @@ -2031,7 +2031,7 @@ static noinline int insert_one_name(struct btrfs_trans_handle *trans, return PTR_ERR(dir); } - ret = btrfs_add_link(trans, dir, inode, name, true, index); + ret = btrfs_add_link(trans, dir, inode, name, true, index, NULL); /* FIXME, put inode into FIXUP list */ @@ -5350,7 +5350,7 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, * have a bunch of extents we just want to commit since it will * be faster. */ - if (++num > 32768) { + if (++num > SZ_16K) { list_del_init(&tree->modified_extents); ret = -EFBIG; goto process; @@ -5368,7 +5368,6 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, refcount_inc(&em->refs); em->flags |= EXTENT_FLAG_LOGGING; list_add_tail(&em->list, &extents); - num++; } list_sort(NULL, &extents, extent_cmp); diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index 600337a84fbe..83dc7dee14cf 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -286,21 +286,17 @@ static int write_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, * @dest: Buffer to read into. This parameter has slightly tricky * semantics. If it is NULL, the function will not do any copying * and will just return the size of all the items up to len bytes. - * If dest_page is passed, then the function will kmap_local the - * page and ignore dest, but it must still be non-NULL to avoid the - * counting-only behavior. * @len: length in bytes to read - * @dest_folio: copy into this folio instead of the dest buffer * * Helper function to read items from the btree. This returns the number of * bytes read or < 0 for errors. We can return short reads if the items don't * exist on disk or aren't big enough to fill the desired length. Supports - * reading into a provided buffer (dest) or into the page cache + * reading into a provided buffer (dest). * * Returns number of bytes read or a negative error code on failure. */ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, - char *dest, u64 len, struct folio *dest_folio) + char *dest, u64 len) { BTRFS_PATH_AUTO_FREE(path); struct btrfs_root *root = inode->root; @@ -320,7 +316,11 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, if (!path) return -ENOMEM; - if (dest_folio) + /* + * Merkle items can be large and split across multiple items, so enable + * readahead for such cases. + */ + if (key_type == BTRFS_VERITY_MERKLE_ITEM_KEY) path->reada = READA_FORWARD; key.objectid = btrfs_ino(inode); @@ -364,7 +364,7 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, break; } - /* desc = NULL to just sum all the item lengths */ + /* dest == NULL to just sum all the item lengths */ if (!dest) copy_end = item_end; else @@ -377,16 +377,10 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, copy_offset = offset - key.offset; if (dest) { - if (dest_folio) - kaddr = kmap_local_folio(dest_folio, 0); - data = btrfs_item_ptr(leaf, path->slots[0], void); read_extent_buffer(leaf, kaddr + dest_offset, (unsigned long)data + copy_offset, copy_bytes); - - if (dest_folio) - kunmap_local(kaddr); } offset += copy_bytes; @@ -677,7 +671,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) memset(&item, 0, sizeof(item)); ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 0, - (char *)&item, sizeof(item), NULL); + (char *)&item, sizeof(item)); if (ret < 0) return ret; @@ -694,7 +688,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) return -ERANGE; ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 1, - buf, buf_size, NULL); + buf, buf_size); if (ret < 0) return ret; if (ret != true_size) @@ -720,6 +714,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, struct folio *folio; u64 off = (u64)index << PAGE_SHIFT; loff_t merkle_pos = merkle_file_pos(inode); + void *kaddr; int ret; if (merkle_pos < 0) @@ -763,6 +758,7 @@ again: } read_folio: + kaddr = kmap_local_folio(folio, 0); /* * Merkle item keys are indexed from byte 0 in the merkle tree. * They have the form: @@ -770,7 +766,8 @@ read_folio: * [ inode objectid, BTRFS_MERKLE_ITEM_KEY, offset in bytes ] */ ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_MERKLE_ITEM_KEY, off, - folio_address(folio), PAGE_SIZE, folio); + kaddr, PAGE_SIZE); + kunmap_local(kaddr); if (ret < 0) { folio_unlock(folio); folio_put(folio); diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 85ea9c5d4536..7c040f22dbc3 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -403,7 +403,7 @@ static struct btrfs_fs_devices *alloc_fs_devices(const u8 *fsid) return fs_devs; } -static void btrfs_free_device(struct btrfs_device *device) +void btrfs_free_device(struct btrfs_device *device) { WARN_ON(!list_empty(&device->post_commit_list)); /* @@ -1358,16 +1358,14 @@ int btrfs_open_devices(struct btrfs_fs_devices *fs_devices, void btrfs_release_disk_super(struct btrfs_super_block *super) { - struct page *page = virt_to_page(super); - - put_page(page); + folio_put(virt_to_folio(super)); } struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, int copy_num, bool drop_cache) { struct btrfs_super_block *super; - struct page *page; + struct folio *folio; u64 bytenr, bytenr_orig; struct address_space *mapping = bdev->bd_mapping; int ret; @@ -1388,7 +1386,7 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, ASSERT(copy_num == 0); /* - * Drop the page of the primary superblock, so later read will + * Drop the folio of the primary superblock, so later read will * always read from the device. */ invalidate_inode_pages2_range(mapping, bytenr >> PAGE_SHIFT, @@ -1396,12 +1394,12 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, } filemap_invalidate_lock_shared(mapping); - page = read_cache_page_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page)) - return ERR_CAST(page); + if (IS_ERR(folio)) + return ERR_CAST(folio); - super = page_address(page); + super = folio_address(folio) + offset_in_folio(folio, bytenr); if (btrfs_super_magic(super) != BTRFS_MAGIC || btrfs_super_bytenr(super) != bytenr_orig) { btrfs_release_disk_super(super); @@ -2814,6 +2812,41 @@ static void btrfs_setup_sprout(struct btrfs_fs_info *fs_info, btrfs_set_super_flags(disk_super, super_flags); } +static void btrfs_rollback_sprout(struct btrfs_fs_info *fs_info, + struct btrfs_fs_devices *seed_devices) +{ + struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; + struct btrfs_super_block *disk_super = fs_info->super_copy; + struct btrfs_device *device; + u64 super_flags; + + lockdep_assert_held(&uuid_mutex); + lockdep_assert_held(&fs_devices->device_list_mutex); + + list_del_init(&seed_devices->seed_list); + list_splice_init_rcu(&seed_devices->devices, &fs_devices->devices, synchronize_rcu); + list_for_each_entry(device, &fs_devices->devices, dev_list) { + device->fs_devices = fs_devices; + } + + fs_devices->seeding = true; + fs_devices->num_devices = seed_devices->num_devices; + fs_devices->open_devices = seed_devices->open_devices; + fs_devices->missing_devices = seed_devices->missing_devices; + fs_devices->rotating = seed_devices->rotating; + fs_devices->latest_dev = seed_devices->latest_dev; + + memcpy(fs_devices->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + memcpy(fs_devices->metadata_uuid, seed_devices->metadata_uuid, BTRFS_FSID_SIZE); + memcpy(disk_super->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + + super_flags = (btrfs_super_flags(disk_super) | BTRFS_SUPER_FLAG_SEEDING); + btrfs_set_super_flags(disk_super, super_flags); + + seed_devices->opened = 0; + free_fs_devices(seed_devices); +} + /* * Store the expected generation for seed devices in device items. */ @@ -3165,6 +3198,8 @@ error_sysfs: orig_super_total_bytes); btrfs_set_super_num_devices(fs_info->super_copy, orig_super_num_devices); + if (seeding_dev) + btrfs_rollback_sprout(fs_info, seed_devices); btrfs_update_per_profile_avail(fs_info); mutex_unlock(&fs_info->chunk_mutex); mutex_unlock(&fs_info->fs_devices->device_list_mutex); diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h index 0415d74cad9b..337d7007d9e2 100644 --- a/fs/btrfs/volumes.h +++ b/fs/btrfs/volumes.h @@ -799,6 +799,7 @@ void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev); void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev); void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev, bool allow_freeze); +void btrfs_free_device(struct btrfs_device *device); unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info, u64 logical); u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map); diff --git a/fs/btrfs/zlib.c b/fs/btrfs/zlib.c index 486b52db583e..e995d0b1452b 100644 --- a/fs/btrfs/zlib.c +++ b/fs/btrfs/zlib.c @@ -49,7 +49,7 @@ void zlib_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->strm.workspace); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -84,13 +84,13 @@ struct list_head *zlib_alloc_workspace(struct btrfs_fs_info *fs_info, unsigned i workspace->level = level; workspace->buf = NULL; if (need_special_buffer(fs_info)) { - workspace->buf = kmalloc(ZLIB_DFLTCC_BUF_SIZE, - __GFP_NOMEMALLOC | __GFP_NORETRY | - __GFP_NOWARN | GFP_NOIO); + workspace->buf = kvmalloc(ZLIB_DFLTCC_BUF_SIZE, + __GFP_NOMEMALLOC | __GFP_NORETRY | + __GFP_NOWARN | GFP_NOIO); workspace->buf_size = ZLIB_DFLTCC_BUF_SIZE; } if (!workspace->buf) { - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); workspace->buf_size = fs_info->sectorsize; } if (!workspace->strm.workspace || !workspace->buf) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 08a15465a087..c04a9955f2d3 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -123,24 +123,24 @@ static int sb_write_pointer(struct block_device *bdev, struct blk_zone *zones, } else if (full[0] && full[1]) { /* Compare two super blocks */ struct address_space *mapping = bdev->bd_mapping; - struct page *page[BTRFS_NR_SB_LOG_ZONES]; struct btrfs_super_block *super[BTRFS_NR_SB_LOG_ZONES]; for (int i = 0; i < BTRFS_NR_SB_LOG_ZONES; i++) { u64 zone_end = (zones[i].start + zones[i].capacity) << SECTOR_SHIFT; u64 bytenr = ALIGN_DOWN(zone_end, BTRFS_SUPER_INFO_SIZE) - BTRFS_SUPER_INFO_SIZE; + struct folio *folio; filemap_invalidate_lock_shared(mapping); - page[i] = read_cache_page_gfp(mapping, - bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, + GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page[i])) { + if (IS_ERR(folio)) { if (i == 1) btrfs_release_disk_super(super[0]); - return PTR_ERR(page[i]); + return PTR_ERR(folio); } - super[i] = page_address(page[i]); + super[i] = folio_address(folio) + offset_in_folio(folio, bytenr); } if (btrfs_super_generation(super[0]) > @@ -804,15 +804,6 @@ int btrfs_check_mountopts_zoned(const struct btrfs_fs_info *info, if (!btrfs_is_zoned(info)) return 0; - /* - * Space cache writing is not COWed. Disable that to avoid write errors - * in sequential zones. - */ - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_err(info, "zoned: space cache v1 is not supported"); - return -EINVAL; - } - if (btrfs_raw_test_opt(*mount_opt, NODATACOW)) { btrfs_err(info, "zoned: NODATACOW not supported"); return -EINVAL; diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 58d9ff76fe07..cc92d0b1b948 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -373,7 +373,7 @@ void zstd_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->mem); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -391,7 +391,7 @@ struct list_head *zstd_alloc_workspace(struct btrfs_fs_info *fs_info, int level) workspace->req_level = level; workspace->last_used = jiffies; workspace->mem = kvmalloc(workspace->size, GFP_KERNEL | __GFP_NOWARN); - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); if (!workspace->mem || !workspace->buf) goto fail; @@ -589,10 +589,48 @@ out: return ret; } +/* + * Map the destination for the next chunk of output. + * + * @decompressed is the offset of the next output byte inside the fully + * decompressed extent. If that offset has reached the current destination + * segment, its page-bounded bio_vec is kmapped so that zstd can write into the + * page cache directly, and the number of bytes writable there is returned. + * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before + * that segment is returned. This covers both the initial prefix and gaps in + * the destination bio. + */ +static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed, + void **kaddr_ret) +{ + struct bio *orig_bio = &cb->orig_bbio->bio; + struct bio_vec bvec; + u32 bvec_offset; + u32 off; + + bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter); + /* + * cb->start may underflow, but subtracting that value can still give us + * the correct offset inside the full decompressed extent. + */ + bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start; + + if (decompressed < bvec_offset) { + *kaddr_ret = NULL; + return bvec_offset - decompressed; + } + + off = decompressed - bvec_offset; + ASSERT(off < bvec.bv_len); + *kaddr_ret = bvec_kmap_local(&bvec) + off; + return bvec.bv_len - off; +} + int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) { struct btrfs_fs_info *fs_info = cb_to_fs_info(cb); struct workspace *workspace = list_entry(ws, struct workspace, list); + struct bio *orig_bio = &cb->orig_bbio->bio; struct folio_iter fi; size_t srclen = bio_get_size(&cb->bbio.bio); zstd_dstream *stream; @@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) const unsigned int min_folio_size = btrfs_min_folio_size(fs_info); unsigned long folio_in_index = 0; unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size); - unsigned long buf_start; unsigned long total_out = 0; bio_first_folio(&fi, &cb->bbio.bio, 0); @@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) workspace->in_buf.pos = 0; workspace->in_buf.size = min_t(size_t, srclen, min_folio_size); - workspace->out_buf.dst = workspace->buf; - workspace->out_buf.pos = 0; - workspace->out_buf.size = fs_info->sectorsize; - - while (1) { + while (orig_bio->bi_iter.bi_size) { size_t ret2; + void *kaddr; + u32 dstlen; + + dstlen = zstd_map_dest(cb, total_out, &kaddr); + if (kaddr) { + workspace->out_buf.dst = kaddr; + workspace->out_buf.size = dstlen; + } else { + workspace->out_buf.dst = workspace->buf; + workspace->out_buf.size = min_t(u32, dstlen, + fs_info->sectorsize); + } + workspace->out_buf.pos = 0; ret2 = zstd_decompress_stream(stream, &workspace->out_buf, &workspace->in_buf); + if (kaddr) + kunmap_local(kaddr); if (unlikely(zstd_is_error(ret2))) { struct btrfs_inode *inode = cb->bbio.inode; @@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) ret = -EIO; goto done; } - buf_start = total_out; total_out += workspace->out_buf.pos; - workspace->out_buf.pos = 0; - - ret = btrfs_decompress_buf2page(workspace->out_buf.dst, - total_out - buf_start, cb, buf_start); - if (ret == 0) - break; + if (kaddr) + bio_advance(orig_bio, workspace->out_buf.pos); if (workspace->in_buf.pos >= srclen) break; |
