summaryrefslogtreecommitdiff
path: root/block
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-04-13 15:51:31 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-04-13 15:51:31 -0700
commit7fe6ac157b7e15c8976bd62ad7cb98e248884e83 (patch)
tree64677a680f3bccc7efb8f4cfcb288006e1433cd3 /block
parentb8f82cb0d84d00c04cdbdce42f67df71b8507e8b (diff)
parent36446de0c30c62b9d89502fd36c4904996d86ecd (diff)
downloadlinux-next-7fe6ac157b7e15c8976bd62ad7cb98e248884e83.tar.gz
linux-next-7fe6ac157b7e15c8976bd62ad7cb98e248884e83.zip
Merge tag 'for-7.1/block-20260411' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux
Pull block updates from Jens Axboe: - Add shared memory zero-copy I/O support for ublk, bypassing per-I/O copies between kernel and userspace by matching registered buffer PFNs at I/O time. Includes selftests. - Refactor bio integrity to support filesystem initiated integrity operations and arbitrary buffer alignment. - Clean up bio allocation, splitting bio_alloc_bioset() into clear fast and slow paths. Add bio_await() and bio_submit_or_kill() helpers, unify synchronous bi_end_io callbacks. - Fix zone write plug refcount handling and plug removal races. Add support for serializing zone writes at QD=1 for rotational zoned devices, yielding significant throughput improvements. - Add SED-OPAL ioctls for Single User Mode management and a STACK_RESET command. - Add io_uring passthrough (uring_cmd) support to the BSG layer. - Replace pp_buf in partition scanning with struct seq_buf. - zloop improvements and cleanups. - drbd genl cleanup, switching to pre_doit/post_doit. - NVMe pull request via Keith: - Fabrics authentication updates - Enhanced block queue limits support - Workqueue usage updates - A new write zeroes device quirk - Tagset cleanup fix for loop device - MD pull requests via Yu Kuai: - Fix raid5 soft lockup in retry_aligned_read() - Fix raid10 deadlock with check operation and nowait requests - Fix raid1 overlapping writes on writemostly disks - Fix sysfs deadlock on array_state=clear - Proactive RAID-5 parity building with llbitmap, with write_zeroes_unmap optimization for initial sync - Fix llbitmap barrier ordering, rdev skipping, and bitmap_ops version mismatch fallback - Fix bcache use-after-free and uninitialized closure - Validate raid5 journal metadata payload size - Various cleanups - Various other fixes, improvements, and cleanups * tag 'for-7.1/block-20260411' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux: (146 commits) ublk: fix tautological comparison warning in ublk_ctrl_reg_buf scsi: bsg: fix buffer overflow in scsi_bsg_uring_cmd() block: refactor blkdev_zone_mgmt_ioctl MAINTAINERS: update ublk driver maintainer email Documentation: ublk: address review comments for SHMEM_ZC docs ublk: allow buffer registration before device is started ublk: replace xarray with IDA for shmem buffer index allocation ublk: simplify PFN range loop in __ublk_ctrl_reg_buf ublk: verify all pages in multi-page bvec fall within registered range ublk: widen ublk_shmem_buf_reg.len to __u64 for 4GB buffer support xfs: use bio_await in xfs_zone_gc_reset_sync block: add a bio_submit_or_kill helper block: factor out a bio_await helper block: unify the synchronous bi_end_io callbacks xfs: fix number of GC bvecs selftests/ublk: add read-only buffer registration test selftests/ublk: add filesystem fio verify test for shmem_zc selftests/ublk: add hugetlbfs shmem_zc test for loop target selftests/ublk: add shared memory zero-copy test selftests/ublk: add UBLK_F_SHMEM_ZC support for loop target ...
Diffstat (limited to 'block')
-rw-r--r--block/bio.c309
-rw-r--r--block/blk-cgroup.c16
-rw-r--r--block/blk-crypto-sysfs.c40
-rw-r--r--block/blk-ia-ranges.c6
-rw-r--r--block/blk-iocost.c23
-rw-r--r--block/blk-lib.c16
-rw-r--r--block/blk-mq-debugfs.c1
-rw-r--r--block/blk-mq-sysfs.c10
-rw-r--r--block/blk-mq.c19
-rw-r--r--block/blk-settings.c12
-rw-r--r--block/blk-sysfs.c89
-rw-r--r--block/blk-wbt.c5
-rw-r--r--block/blk-zoned.c469
-rw-r--r--block/blk.h7
-rw-r--r--block/bsg-lib.c2
-rw-r--r--block/bsg.c33
-rw-r--r--block/disk-events.c3
-rw-r--r--block/ioctl.c11
-rw-r--r--block/opal_proto.h24
-rw-r--r--block/partitions/acorn.c32
-rw-r--r--block/partitions/aix.c21
-rw-r--r--block/partitions/amiga.c35
-rw-r--r--block/partitions/atari.c12
-rw-r--r--block/partitions/check.h8
-rw-r--r--block/partitions/cmdline.c6
-rw-r--r--block/partitions/core.c31
-rw-r--r--block/partitions/efi.c2
-rw-r--r--block/partitions/ibm.c27
-rw-r--r--block/partitions/karma.c2
-rw-r--r--block/partitions/ldm.c4
-rw-r--r--block/partitions/mac.c4
-rw-r--r--block/partitions/msdos.c67
-rw-r--r--block/partitions/of.c6
-rw-r--r--block/partitions/osf.c2
-rw-r--r--block/partitions/sgi.c2
-rw-r--r--block/partitions/sun.c2
-rw-r--r--block/partitions/sysv68.c9
-rw-r--r--block/partitions/ultrix.c2
-rw-r--r--block/sed-opal.c434
-rw-r--r--block/t10-pi.c816
40 files changed, 1638 insertions, 981 deletions
diff --git a/block/bio.c b/block/bio.c
index 784d2a66d3ae..641ef0928d73 100644
--- a/block/bio.c
+++ b/block/bio.c
@@ -18,6 +18,7 @@
#include <linux/highmem.h>
#include <linux/blk-crypto.h>
#include <linux/xarray.h>
+#include <linux/kmemleak.h>
#include <trace/events/block.h>
#include "blk.h"
@@ -34,6 +35,8 @@ struct bio_alloc_cache {
unsigned int nr_irq;
};
+#define BIO_INLINE_VECS 4
+
static struct biovec_slab {
int nr_vecs;
char *name;
@@ -114,6 +117,11 @@ static inline unsigned int bs_bio_slab_size(struct bio_set *bs)
return bs->front_pad + sizeof(struct bio) + bs->back_pad;
}
+static inline void *bio_slab_addr(struct bio *bio)
+{
+ return (void *)bio - bio->bi_pool->front_pad;
+}
+
static struct kmem_cache *bio_find_or_create_slab(struct bio_set *bs)
{
unsigned int size = bs_bio_slab_size(bs);
@@ -159,57 +167,16 @@ out:
mutex_unlock(&bio_slab_lock);
}
-void bvec_free(mempool_t *pool, struct bio_vec *bv, unsigned short nr_vecs)
-{
- BUG_ON(nr_vecs > BIO_MAX_VECS);
-
- if (nr_vecs == BIO_MAX_VECS)
- mempool_free(bv, pool);
- else if (nr_vecs > BIO_INLINE_VECS)
- kmem_cache_free(biovec_slab(nr_vecs)->slab, bv);
-}
-
/*
* Make the first allocation restricted and don't dump info on allocation
* failures, since we'll fall back to the mempool in case of failure.
*/
-static inline gfp_t bvec_alloc_gfp(gfp_t gfp)
+static inline gfp_t try_alloc_gfp(gfp_t gfp)
{
return (gfp & ~(__GFP_DIRECT_RECLAIM | __GFP_IO)) |
__GFP_NOMEMALLOC | __GFP_NORETRY | __GFP_NOWARN;
}
-struct bio_vec *bvec_alloc(mempool_t *pool, unsigned short *nr_vecs,
- gfp_t gfp_mask)
-{
- struct biovec_slab *bvs = biovec_slab(*nr_vecs);
-
- if (WARN_ON_ONCE(!bvs))
- return NULL;
-
- /*
- * Upgrade the nr_vecs request to take full advantage of the allocation.
- * We also rely on this in the bvec_free path.
- */
- *nr_vecs = bvs->nr_vecs;
-
- /*
- * Try a slab allocation first for all smaller allocations. If that
- * fails and __GFP_DIRECT_RECLAIM is set retry with the mempool.
- * The mempool is sized to handle up to BIO_MAX_VECS entries.
- */
- if (*nr_vecs < BIO_MAX_VECS) {
- struct bio_vec *bvl;
-
- bvl = kmem_cache_alloc(bvs->slab, bvec_alloc_gfp(gfp_mask));
- if (likely(bvl) || !(gfp_mask & __GFP_DIRECT_RECLAIM))
- return bvl;
- *nr_vecs = BIO_MAX_VECS;
- }
-
- return mempool_alloc(pool, gfp_mask);
-}
-
void bio_uninit(struct bio *bio)
{
#ifdef CONFIG_BLK_CGROUP
@@ -231,9 +198,14 @@ static void bio_free(struct bio *bio)
void *p = bio;
WARN_ON_ONCE(!bs);
+ WARN_ON_ONCE(bio->bi_max_vecs > BIO_MAX_VECS);
bio_uninit(bio);
- bvec_free(&bs->bvec_pool, bio->bi_io_vec, bio->bi_max_vecs);
+ if (bio->bi_max_vecs == BIO_MAX_VECS)
+ mempool_free(bio->bi_io_vec, &bs->bvec_pool);
+ else if (bio->bi_max_vecs > BIO_INLINE_VECS)
+ kmem_cache_free(biovec_slab(bio->bi_max_vecs)->slab,
+ bio->bi_io_vec);
mempool_free(p - bs->front_pad, &bs->bio_pool);
}
@@ -430,13 +402,31 @@ static void bio_alloc_rescue(struct work_struct *work)
}
}
+/*
+ * submit_bio_noacct() converts recursion to iteration; this means if we're
+ * running beneath it, any bios we allocate and submit will not be submitted
+ * (and thus freed) until after we return.
+ *
+ * This exposes us to a potential deadlock if we allocate multiple bios from the
+ * same bio_set while running underneath submit_bio_noacct(). If we were to
+ * allocate multiple bios (say a stacking block driver that was splitting bios),
+ * we would deadlock if we exhausted the mempool's reserve.
+ *
+ * We solve this, and guarantee forward progress by punting the bios on
+ * current->bio_list to a per bio_set rescuer workqueue before blocking to wait
+ * for elements being returned to the mempool.
+ */
static void punt_bios_to_rescuer(struct bio_set *bs)
{
struct bio_list punt, nopunt;
struct bio *bio;
- if (WARN_ON_ONCE(!bs->rescue_workqueue))
+ if (!current->bio_list || !bs->rescue_workqueue)
+ return;
+ if (bio_list_empty(&current->bio_list[0]) &&
+ bio_list_empty(&current->bio_list[1]))
return;
+
/*
* In order to guarantee forward progress we must punt only bios that
* were allocated from this bio_set; otherwise, if there was a bio on
@@ -483,9 +473,7 @@ static void bio_alloc_irq_cache_splice(struct bio_alloc_cache *cache)
local_irq_restore(flags);
}
-static struct bio *bio_alloc_percpu_cache(struct block_device *bdev,
- unsigned short nr_vecs, blk_opf_t opf, gfp_t gfp,
- struct bio_set *bs)
+static struct bio *bio_alloc_percpu_cache(struct bio_set *bs)
{
struct bio_alloc_cache *cache;
struct bio *bio;
@@ -503,12 +491,10 @@ static struct bio *bio_alloc_percpu_cache(struct block_device *bdev,
cache->free_list = bio->bi_next;
cache->nr--;
put_cpu();
-
- if (nr_vecs)
- bio_init_inline(bio, bdev, nr_vecs, opf);
- else
- bio_init(bio, bdev, NULL, nr_vecs, opf);
bio->bi_pool = bs;
+
+ kmemleak_alloc(bio_slab_addr(bio),
+ kmem_cache_size(bs->bio_slab), 1, GFP_NOIO);
return bio;
}
@@ -517,7 +503,7 @@ static struct bio *bio_alloc_percpu_cache(struct block_device *bdev,
* @bdev: block device to allocate the bio for (can be %NULL)
* @nr_vecs: number of bvecs to pre-allocate
* @opf: operation and flags for bio
- * @gfp_mask: the GFP_* mask given to the slab allocator
+ * @gfp: the GFP_* mask given to the slab allocator
* @bs: the bio_set to allocate from.
*
* Allocate a bio from the mempools in @bs.
@@ -547,91 +533,77 @@ static struct bio *bio_alloc_percpu_cache(struct block_device *bdev,
* Returns: Pointer to new bio on success, NULL on failure.
*/
struct bio *bio_alloc_bioset(struct block_device *bdev, unsigned short nr_vecs,
- blk_opf_t opf, gfp_t gfp_mask,
- struct bio_set *bs)
+ blk_opf_t opf, gfp_t gfp, struct bio_set *bs)
{
- gfp_t saved_gfp = gfp_mask;
- struct bio *bio;
+ struct bio_vec *bvecs = NULL;
+ struct bio *bio = NULL;
+ gfp_t saved_gfp = gfp;
void *p;
/* should not use nobvec bioset for nr_vecs > 0 */
if (WARN_ON_ONCE(!mempool_initialized(&bs->bvec_pool) && nr_vecs > 0))
return NULL;
+ gfp = try_alloc_gfp(gfp);
if (bs->cache && nr_vecs <= BIO_INLINE_VECS) {
- opf |= REQ_ALLOC_CACHE;
- bio = bio_alloc_percpu_cache(bdev, nr_vecs, opf,
- gfp_mask, bs);
- if (bio)
- return bio;
/*
- * No cached bio available, bio returned below marked with
- * REQ_ALLOC_CACHE to participate in per-cpu alloc cache.
+ * Set REQ_ALLOC_CACHE even if no cached bio is available to
+ * return the allocated bio to the percpu cache when done.
*/
- } else
+ opf |= REQ_ALLOC_CACHE;
+ bio = bio_alloc_percpu_cache(bs);
+ } else {
opf &= ~REQ_ALLOC_CACHE;
-
- /*
- * submit_bio_noacct() converts recursion to iteration; this means if
- * we're running beneath it, any bios we allocate and submit will not be
- * submitted (and thus freed) until after we return.
- *
- * This exposes us to a potential deadlock if we allocate multiple bios
- * from the same bio_set() while running underneath submit_bio_noacct().
- * If we were to allocate multiple bios (say a stacking block driver
- * that was splitting bios), we would deadlock if we exhausted the
- * mempool's reserve.
- *
- * We solve this, and guarantee forward progress, with a rescuer
- * workqueue per bio_set. If we go to allocate and there are bios on
- * current->bio_list, we first try the allocation without
- * __GFP_DIRECT_RECLAIM; if that fails, we punt those bios we would be
- * blocking to the rescuer workqueue before we retry with the original
- * gfp_flags.
- */
- if (current->bio_list &&
- (!bio_list_empty(&current->bio_list[0]) ||
- !bio_list_empty(&current->bio_list[1])) &&
- bs->rescue_workqueue)
- gfp_mask &= ~__GFP_DIRECT_RECLAIM;
-
- p = mempool_alloc(&bs->bio_pool, gfp_mask);
- if (!p && gfp_mask != saved_gfp) {
- punt_bios_to_rescuer(bs);
- gfp_mask = saved_gfp;
- p = mempool_alloc(&bs->bio_pool, gfp_mask);
+ p = kmem_cache_alloc(bs->bio_slab, gfp);
+ if (p)
+ bio = p + bs->front_pad;
}
- if (unlikely(!p))
- return NULL;
- if (!mempool_is_saturated(&bs->bio_pool))
- opf &= ~REQ_ALLOC_CACHE;
- bio = p + bs->front_pad;
- if (nr_vecs > BIO_INLINE_VECS) {
- struct bio_vec *bvl = NULL;
+ if (bio && nr_vecs > BIO_INLINE_VECS) {
+ struct biovec_slab *bvs = biovec_slab(nr_vecs);
- bvl = bvec_alloc(&bs->bvec_pool, &nr_vecs, gfp_mask);
- if (!bvl && gfp_mask != saved_gfp) {
- punt_bios_to_rescuer(bs);
- gfp_mask = saved_gfp;
- bvl = bvec_alloc(&bs->bvec_pool, &nr_vecs, gfp_mask);
+ /*
+ * Upgrade nr_vecs to take full advantage of the allocation.
+ * We also rely on this in bio_free().
+ */
+ nr_vecs = bvs->nr_vecs;
+ bvecs = kmem_cache_alloc(bvs->slab, gfp);
+ if (unlikely(!bvecs)) {
+ kmem_cache_free(bs->bio_slab, p);
+ bio = NULL;
}
- if (unlikely(!bvl))
- goto err_free;
+ }
- bio_init(bio, bdev, bvl, nr_vecs, opf);
- } else if (nr_vecs) {
- bio_init_inline(bio, bdev, BIO_INLINE_VECS, opf);
- } else {
- bio_init(bio, bdev, NULL, 0, opf);
+ if (unlikely(!bio)) {
+ /*
+ * Give up if we are not allow to sleep as non-blocking mempool
+ * allocations just go back to the slab allocation.
+ */
+ if (!(saved_gfp & __GFP_DIRECT_RECLAIM))
+ return NULL;
+
+ punt_bios_to_rescuer(bs);
+
+ /*
+ * Don't rob the mempools by returning to the per-CPU cache if
+ * we're tight on memory.
+ */
+ opf &= ~REQ_ALLOC_CACHE;
+
+ p = mempool_alloc(&bs->bio_pool, saved_gfp);
+ bio = p + bs->front_pad;
+ if (nr_vecs > BIO_INLINE_VECS) {
+ nr_vecs = BIO_MAX_VECS;
+ bvecs = mempool_alloc(&bs->bvec_pool, saved_gfp);
+ }
}
+ if (nr_vecs && nr_vecs <= BIO_INLINE_VECS)
+ bio_init_inline(bio, bdev, nr_vecs, opf);
+ else
+ bio_init(bio, bdev, bvecs, nr_vecs, opf);
bio->bi_pool = bs;
return bio;
-
-err_free:
- mempool_free(p, &bs->bio_pool);
- return NULL;
}
EXPORT_SYMBOL(bio_alloc_bioset);
@@ -765,6 +737,9 @@ static int __bio_alloc_cache_prune(struct bio_alloc_cache *cache,
while ((bio = cache->free_list) != NULL) {
cache->free_list = bio->bi_next;
cache->nr--;
+ kmemleak_alloc(bio_slab_addr(bio),
+ kmem_cache_size(bio->bi_pool->bio_slab),
+ 1, GFP_KERNEL);
bio_free(bio);
if (++i == nr)
break;
@@ -828,6 +803,7 @@ static inline void bio_put_percpu_cache(struct bio *bio)
bio->bi_bdev = NULL;
cache->free_list = bio;
cache->nr++;
+ kmemleak_free(bio_slab_addr(bio));
} else if (in_hardirq()) {
lockdep_assert_irqs_disabled();
@@ -835,6 +811,7 @@ static inline void bio_put_percpu_cache(struct bio *bio)
bio->bi_next = cache->free_list_irq;
cache->free_list_irq = bio;
cache->nr_irq++;
+ kmemleak_free(bio_slab_addr(bio));
} else {
goto out_free;
}
@@ -897,10 +874,11 @@ static int __bio_clone(struct bio *bio, struct bio *bio_src, gfp_t gfp)
* @gfp: allocation priority
* @bs: bio_set to allocate from
*
- * Allocate a new bio that is a clone of @bio_src. The caller owns the returned
- * bio, but not the actual data it points to.
- *
- * The caller must ensure that the return bio is not freed before @bio_src.
+ * Allocate a new bio that is a clone of @bio_src. This reuses the bio_vecs
+ * pointed to by @bio_src->bi_io_vec, and clones the iterator pointing to
+ * the current position in it. The caller owns the returned bio, but not
+ * the bio_vecs, and must ensure the bio is freed before the memory
+ * pointed to by @bio_Src->bi_io_vecs.
*/
struct bio *bio_alloc_clone(struct block_device *bdev, struct bio *bio_src,
gfp_t gfp, struct bio_set *bs)
@@ -929,9 +907,7 @@ EXPORT_SYMBOL(bio_alloc_clone);
* @gfp: allocation priority
*
* Initialize a new bio in caller provided memory that is a clone of @bio_src.
- * The caller owns the returned bio, but not the actual data it points to.
- *
- * The caller must ensure that @bio_src is not freed before @bio.
+ * The same bio_vecs reuse and bio lifetime rules as bio_alloc_clone() apply.
*/
int bio_init_clone(struct block_device *bdev, struct bio *bio,
struct bio *bio_src, gfp_t gfp)
@@ -1064,6 +1040,8 @@ int bio_add_page(struct bio *bio, struct page *page,
{
if (WARN_ON_ONCE(bio_flagged(bio, BIO_CLONED)))
return 0;
+ if (WARN_ON_ONCE(len == 0))
+ return 0;
if (bio->bi_iter.bi_size > BIO_MAX_SIZE - len)
return 0;
@@ -1484,12 +1462,42 @@ void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty)
bio_iov_iter_unbounce_read(bio, is_error, mark_dirty);
}
-static void submit_bio_wait_endio(struct bio *bio)
+static void bio_wait_end_io(struct bio *bio)
{
complete(bio->bi_private);
}
/**
+ * bio_await - call a function on a bio, and wait until it completes
+ * @bio: the bio which describes the I/O
+ * @submit: function called to submit the bio
+ * @priv: private data passed to @submit
+ *
+ * Wait for the bio as well as any bio chained off it after executing the
+ * passed in callback @submit. The wait for the bio is set up before calling
+ * @submit to ensure that the completion is captured. If @submit is %NULL,
+ * submit_bio() is used instead to submit the bio.
+ *
+ * Note: this overrides the bi_private and bi_end_io fields in the bio.
+ */
+void bio_await(struct bio *bio, void *priv,
+ void (*submit)(struct bio *bio, void *priv))
+{
+ DECLARE_COMPLETION_ONSTACK_MAP(done,
+ bio->bi_bdev->bd_disk->lockdep_map);
+
+ bio->bi_private = &done;
+ bio->bi_end_io = bio_wait_end_io;
+ bio->bi_opf |= REQ_SYNC;
+ if (submit)
+ submit(bio, priv);
+ else
+ submit_bio(bio);
+ blk_wait_io(&done);
+}
+EXPORT_SYMBOL_GPL(bio_await);
+
+/**
* submit_bio_wait - submit a bio, and wait until it completes
* @bio: The &struct bio which describes the I/O
*
@@ -1502,19 +1510,30 @@ static void submit_bio_wait_endio(struct bio *bio)
*/
int submit_bio_wait(struct bio *bio)
{
- DECLARE_COMPLETION_ONSTACK_MAP(done,
- bio->bi_bdev->bd_disk->lockdep_map);
-
- bio->bi_private = &done;
- bio->bi_end_io = submit_bio_wait_endio;
- bio->bi_opf |= REQ_SYNC;
- submit_bio(bio);
- blk_wait_io(&done);
-
+ bio_await(bio, NULL, NULL);
return blk_status_to_errno(bio->bi_status);
}
EXPORT_SYMBOL(submit_bio_wait);
+static void bio_endio_cb(struct bio *bio, void *priv)
+{
+ bio_endio(bio);
+}
+
+/*
+ * Submit @bio synchronously, or call bio_endio on it if the current process
+ * is being killed.
+ */
+int bio_submit_or_kill(struct bio *bio, unsigned int flags)
+{
+ if ((flags & BLKDEV_ZERO_KILLABLE) && fatal_signal_pending(current)) {
+ bio_await(bio, NULL, bio_endio_cb);
+ return -EINTR;
+ }
+
+ return submit_bio_wait(bio);
+}
+
/**
* bdev_rw_virt - synchronously read into / write from kernel mapping
* @bdev: block device to access
@@ -1545,26 +1564,6 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data,
}
EXPORT_SYMBOL_GPL(bdev_rw_virt);
-static void bio_wait_end_io(struct bio *bio)
-{
- complete(bio->bi_private);
- bio_put(bio);
-}
-
-/*
- * bio_await_chain - ends @bio and waits for every chained bio to complete
- */
-void bio_await_chain(struct bio *bio)
-{
- DECLARE_COMPLETION_ONSTACK_MAP(done,
- bio->bi_bdev->bd_disk->lockdep_map);
-
- bio->bi_private = &done;
- bio->bi_end_io = bio_wait_end_io;
- bio_endio(bio);
- blk_wait_io(&done);
-}
-
void __bio_advance(struct bio *bio, unsigned bytes)
{
if (bio_integrity(bio))
diff --git a/block/blk-cgroup.c b/block/blk-cgroup.c
index b70096497d38..554c87bb4a86 100644
--- a/block/blk-cgroup.c
+++ b/block/blk-cgroup.c
@@ -24,6 +24,7 @@
#include <linux/backing-dev.h>
#include <linux/slab.h>
#include <linux/delay.h>
+#include <linux/wait_bit.h>
#include <linux/atomic.h>
#include <linux/ctype.h>
#include <linux/resume_user_mode.h>
@@ -611,6 +612,8 @@ restart:
q->root_blkg = NULL;
spin_unlock_irq(&q->queue_lock);
+
+ wake_up_var(&q->root_blkg);
}
static void blkg_iostat_set(struct blkg_iostat *dst, struct blkg_iostat *src)
@@ -1498,6 +1501,18 @@ int blkcg_init_disk(struct gendisk *disk)
struct blkcg_gq *new_blkg, *blkg;
bool preloaded;
+ /*
+ * If the queue is shared across disk rebind (e.g., SCSI), the
+ * previous disk's blkcg state is cleaned up asynchronously via
+ * disk_release() -> blkcg_exit_disk(). Wait for that cleanup to
+ * finish (indicated by root_blkg becoming NULL) before setting up
+ * new blkcg state. Otherwise, we may overwrite q->root_blkg while
+ * the old one is still alive, and radix_tree_insert() in
+ * blkg_create() will fail with -EEXIST because the old entries
+ * still occupy the same queue id slot in blkcg->blkg_tree.
+ */
+ wait_var_event(&q->root_blkg, !READ_ONCE(q->root_blkg));
+
new_blkg = blkg_alloc(&blkcg_root, disk, GFP_KERNEL);
if (!new_blkg)
return -ENOMEM;
@@ -2022,6 +2037,7 @@ void blkcg_maybe_throttle_current(void)
return;
out:
rcu_read_unlock();
+ put_disk(disk);
}
/**
diff --git a/block/blk-crypto-sysfs.c b/block/blk-crypto-sysfs.c
index ea7a0b85a46f..b069c418b6cc 100644
--- a/block/blk-crypto-sysfs.c
+++ b/block/blk-crypto-sysfs.c
@@ -18,7 +18,7 @@ struct blk_crypto_kobj {
struct blk_crypto_attr {
struct attribute attr;
ssize_t (*show)(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page);
+ const struct blk_crypto_attr *attr, char *page);
};
static struct blk_crypto_profile *kobj_to_crypto_profile(struct kobject *kobj)
@@ -26,39 +26,39 @@ static struct blk_crypto_profile *kobj_to_crypto_profile(struct kobject *kobj)
return container_of(kobj, struct blk_crypto_kobj, kobj)->profile;
}
-static struct blk_crypto_attr *attr_to_crypto_attr(struct attribute *attr)
+static const struct blk_crypto_attr *attr_to_crypto_attr(const struct attribute *attr)
{
- return container_of(attr, struct blk_crypto_attr, attr);
+ return container_of_const(attr, struct blk_crypto_attr, attr);
}
static ssize_t hw_wrapped_keys_show(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page)
+ const struct blk_crypto_attr *attr, char *page)
{
/* Always show supported, since the file doesn't exist otherwise. */
return sysfs_emit(page, "supported\n");
}
static ssize_t max_dun_bits_show(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page)
+ const struct blk_crypto_attr *attr, char *page)
{
return sysfs_emit(page, "%u\n", 8 * profile->max_dun_bytes_supported);
}
static ssize_t num_keyslots_show(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page)
+ const struct blk_crypto_attr *attr, char *page)
{
return sysfs_emit(page, "%u\n", profile->num_slots);
}
static ssize_t raw_keys_show(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page)
+ const struct blk_crypto_attr *attr, char *page)
{
/* Always show supported, since the file doesn't exist otherwise. */
return sysfs_emit(page, "supported\n");
}
#define BLK_CRYPTO_RO_ATTR(_name) \
- static struct blk_crypto_attr _name##_attr = __ATTR_RO(_name)
+ static const struct blk_crypto_attr _name##_attr = __ATTR_RO(_name)
BLK_CRYPTO_RO_ATTR(hw_wrapped_keys);
BLK_CRYPTO_RO_ATTR(max_dun_bits);
@@ -66,10 +66,10 @@ BLK_CRYPTO_RO_ATTR(num_keyslots);
BLK_CRYPTO_RO_ATTR(raw_keys);
static umode_t blk_crypto_is_visible(struct kobject *kobj,
- struct attribute *attr, int n)
+ const struct attribute *attr, int n)
{
struct blk_crypto_profile *profile = kobj_to_crypto_profile(kobj);
- struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
+ const struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
if (a == &hw_wrapped_keys_attr &&
!(profile->key_types_supported & BLK_CRYPTO_KEY_TYPE_HW_WRAPPED))
@@ -81,7 +81,7 @@ static umode_t blk_crypto_is_visible(struct kobject *kobj,
return 0444;
}
-static struct attribute *blk_crypto_attrs[] = {
+static const struct attribute *const blk_crypto_attrs[] = {
&hw_wrapped_keys_attr.attr,
&max_dun_bits_attr.attr,
&num_keyslots_attr.attr,
@@ -90,8 +90,8 @@ static struct attribute *blk_crypto_attrs[] = {
};
static const struct attribute_group blk_crypto_attr_group = {
- .attrs = blk_crypto_attrs,
- .is_visible = blk_crypto_is_visible,
+ .attrs_const = blk_crypto_attrs,
+ .is_visible_const = blk_crypto_is_visible,
};
/*
@@ -99,13 +99,13 @@ static const struct attribute_group blk_crypto_attr_group = {
* modes, these are initialized at boot time by blk_crypto_sysfs_init().
*/
static struct blk_crypto_attr __blk_crypto_mode_attrs[BLK_ENCRYPTION_MODE_MAX];
-static struct attribute *blk_crypto_mode_attrs[BLK_ENCRYPTION_MODE_MAX + 1];
+static const struct attribute *blk_crypto_mode_attrs[BLK_ENCRYPTION_MODE_MAX + 1];
static umode_t blk_crypto_mode_is_visible(struct kobject *kobj,
- struct attribute *attr, int n)
+ const struct attribute *attr, int n)
{
struct blk_crypto_profile *profile = kobj_to_crypto_profile(kobj);
- struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
+ const struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
int mode_num = a - __blk_crypto_mode_attrs;
if (profile->modes_supported[mode_num])
@@ -114,7 +114,7 @@ static umode_t blk_crypto_mode_is_visible(struct kobject *kobj,
}
static ssize_t blk_crypto_mode_show(struct blk_crypto_profile *profile,
- struct blk_crypto_attr *attr, char *page)
+ const struct blk_crypto_attr *attr, char *page)
{
int mode_num = attr - __blk_crypto_mode_attrs;
@@ -123,8 +123,8 @@ static ssize_t blk_crypto_mode_show(struct blk_crypto_profile *profile,
static const struct attribute_group blk_crypto_modes_attr_group = {
.name = "modes",
- .attrs = blk_crypto_mode_attrs,
- .is_visible = blk_crypto_mode_is_visible,
+ .attrs_const = blk_crypto_mode_attrs,
+ .is_visible_const = blk_crypto_mode_is_visible,
};
static const struct attribute_group *blk_crypto_attr_groups[] = {
@@ -137,7 +137,7 @@ static ssize_t blk_crypto_attr_show(struct kobject *kobj,
struct attribute *attr, char *page)
{
struct blk_crypto_profile *profile = kobj_to_crypto_profile(kobj);
- struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
+ const struct blk_crypto_attr *a = attr_to_crypto_attr(attr);
return a->show(profile, a, page);
}
diff --git a/block/blk-ia-ranges.c b/block/blk-ia-ranges.c
index d479f5481b66..7be8b58893c9 100644
--- a/block/blk-ia-ranges.c
+++ b/block/blk-ia-ranges.c
@@ -30,17 +30,17 @@ struct blk_ia_range_sysfs_entry {
ssize_t (*show)(struct blk_independent_access_range *iar, char *buf);
};
-static struct blk_ia_range_sysfs_entry blk_ia_range_sector_entry = {
+static const struct blk_ia_range_sysfs_entry blk_ia_range_sector_entry = {
.attr = { .name = "sector", .mode = 0444 },
.show = blk_ia_range_sector_show,
};
-static struct blk_ia_range_sysfs_entry blk_ia_range_nr_sectors_entry = {
+static const struct blk_ia_range_sysfs_entry blk_ia_range_nr_sectors_entry = {
.attr = { .name = "nr_sectors", .mode = 0444 },
.show = blk_ia_range_nr_sectors_show,
};
-static struct attribute *blk_ia_range_attrs[] = {
+static const struct attribute *const blk_ia_range_attrs[] = {
&blk_ia_range_sector_entry.attr,
&blk_ia_range_nr_sectors_entry.attr,
NULL,
diff --git a/block/blk-iocost.c b/block/blk-iocost.c
index d145db61e5c3..0cca88a366dc 100644
--- a/block/blk-iocost.c
+++ b/block/blk-iocost.c
@@ -1596,7 +1596,8 @@ static enum hrtimer_restart iocg_waitq_timer_fn(struct hrtimer *timer)
return HRTIMER_NORESTART;
}
-static void ioc_lat_stat(struct ioc *ioc, u32 *missed_ppm_ar, u32 *rq_wait_pct_p)
+static void ioc_lat_stat(struct ioc *ioc, u32 *missed_ppm_ar, u32 *rq_wait_pct_p,
+ u32 *nr_done)
{
u32 nr_met[2] = { };
u32 nr_missed[2] = { };
@@ -1633,6 +1634,8 @@ static void ioc_lat_stat(struct ioc *ioc, u32 *missed_ppm_ar, u32 *rq_wait_pct_p
*rq_wait_pct_p = div64_u64(rq_wait_ns * 100,
ioc->period_us * NSEC_PER_USEC);
+
+ *nr_done = nr_met[READ] + nr_met[WRITE] + nr_missed[READ] + nr_missed[WRITE];
}
/* was iocg idle this period? */
@@ -2250,12 +2253,12 @@ static void ioc_timer_fn(struct timer_list *timer)
u64 usage_us_sum = 0;
u32 ppm_rthr;
u32 ppm_wthr;
- u32 missed_ppm[2], rq_wait_pct;
+ u32 missed_ppm[2], rq_wait_pct, nr_done;
u64 period_vtime;
int prev_busy_level;
/* how were the latencies during the period? */
- ioc_lat_stat(ioc, missed_ppm, &rq_wait_pct);
+ ioc_lat_stat(ioc, missed_ppm, &rq_wait_pct, &nr_done);
/* take care of active iocgs */
spin_lock_irq(&ioc->lock);
@@ -2397,9 +2400,17 @@ static void ioc_timer_fn(struct timer_list *timer)
* and should increase vtime rate.
*/
prev_busy_level = ioc->busy_level;
- if (rq_wait_pct > RQ_WAIT_BUSY_PCT ||
- missed_ppm[READ] > ppm_rthr ||
- missed_ppm[WRITE] > ppm_wthr) {
+ if (!nr_done && nr_lagging) {
+ /*
+ * When there are lagging IOs but no completions, we don't
+ * know if the IO latency will meet the QoS targets. The
+ * disk might be saturated or not. We should not reset
+ * busy_level to 0 (which would prevent vrate from scaling
+ * up or down), but rather to keep it unchanged.
+ */
+ } else if (rq_wait_pct > RQ_WAIT_BUSY_PCT ||
+ missed_ppm[READ] > ppm_rthr ||
+ missed_ppm[WRITE] > ppm_wthr) {
/* clearly missing QoS targets, slow down vrate */
ioc->busy_level = max(ioc->busy_level, 0);
ioc->busy_level++;
diff --git a/block/blk-lib.c b/block/blk-lib.c
index 3213afc7f0d5..688bc67cbf73 100644
--- a/block/blk-lib.c
+++ b/block/blk-lib.c
@@ -155,13 +155,7 @@ static int blkdev_issue_write_zeroes(struct block_device *bdev, sector_t sector,
__blkdev_issue_write_zeroes(bdev, sector, nr_sects, gfp, &bio,
flags, limit);
if (bio) {
- if ((flags & BLKDEV_ZERO_KILLABLE) &&
- fatal_signal_pending(current)) {
- bio_await_chain(bio);
- blk_finish_plug(&plug);
- return -EINTR;
- }
- ret = submit_bio_wait(bio);
+ ret = bio_submit_or_kill(bio, flags);
bio_put(bio);
}
blk_finish_plug(&plug);
@@ -236,13 +230,7 @@ static int blkdev_issue_zero_pages(struct block_device *bdev, sector_t sector,
blk_start_plug(&plug);
__blkdev_issue_zero_pages(bdev, sector, nr_sects, gfp, &bio, flags);
if (bio) {
- if ((flags & BLKDEV_ZERO_KILLABLE) &&
- fatal_signal_pending(current)) {
- bio_await_chain(bio);
- blk_finish_plug(&plug);
- return -EINTR;
- }
- ret = submit_bio_wait(bio);
+ ret = bio_submit_or_kill(bio, flags);
bio_put(bio);
}
blk_finish_plug(&plug);
diff --git a/block/blk-mq-debugfs.c b/block/blk-mq-debugfs.c
index 28167c9baa55..047ec887456b 100644
--- a/block/blk-mq-debugfs.c
+++ b/block/blk-mq-debugfs.c
@@ -97,6 +97,7 @@ static const char *const blk_queue_flag_name[] = {
QUEUE_FLAG_NAME(NO_ELV_SWITCH),
QUEUE_FLAG_NAME(QOS_ENABLED),
QUEUE_FLAG_NAME(BIO_ISSUE_TIME),
+ QUEUE_FLAG_NAME(ZONED_QD1_WRITES),
};
#undef QUEUE_FLAG_NAME
diff --git a/block/blk-mq-sysfs.c b/block/blk-mq-sysfs.c
index 58ec293373c6..895397831ecc 100644
--- a/block/blk-mq-sysfs.c
+++ b/block/blk-mq-sysfs.c
@@ -53,7 +53,7 @@ static ssize_t blk_mq_hw_sysfs_show(struct kobject *kobj,
struct request_queue *q;
ssize_t res;
- entry = container_of(attr, struct blk_mq_hw_ctx_sysfs_entry, attr);
+ entry = container_of_const(attr, struct blk_mq_hw_ctx_sysfs_entry, attr);
hctx = container_of(kobj, struct blk_mq_hw_ctx, kobj);
q = hctx->queue;
@@ -101,20 +101,20 @@ static ssize_t blk_mq_hw_sysfs_cpus_show(struct blk_mq_hw_ctx *hctx, char *page)
return pos + ret;
}
-static struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_nr_tags = {
+static const struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_nr_tags = {
.attr = {.name = "nr_tags", .mode = 0444 },
.show = blk_mq_hw_sysfs_nr_tags_show,
};
-static struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_nr_reserved_tags = {
+static const struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_nr_reserved_tags = {
.attr = {.name = "nr_reserved_tags", .mode = 0444 },
.show = blk_mq_hw_sysfs_nr_reserved_tags_show,
};
-static struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_cpus = {
+static const struct blk_mq_hw_ctx_sysfs_entry blk_mq_hw_sysfs_cpus = {
.attr = {.name = "cpu_list", .mode = 0444 },
.show = blk_mq_hw_sysfs_cpus_show,
};
-static struct attribute *default_hw_ctx_attrs[] = {
+static const struct attribute *const default_hw_ctx_attrs[] = {
&blk_mq_hw_sysfs_nr_tags.attr,
&blk_mq_hw_sysfs_nr_reserved_tags.attr,
&blk_mq_hw_sysfs_cpus.attr,
diff --git a/block/blk-mq.c b/block/blk-mq.c
index a047faf3b0ec..4c5c16cce4f8 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -3424,6 +3424,25 @@ EXPORT_SYMBOL_GPL(blk_rq_prep_clone);
*/
void blk_steal_bios(struct bio_list *list, struct request *rq)
{
+ struct bio *bio;
+
+ for (bio = rq->bio; bio; bio = bio->bi_next) {
+ if (bio->bi_opf & REQ_POLLED) {
+ bio->bi_opf &= ~REQ_POLLED;
+ bio->bi_cookie = BLK_QC_T_NONE;
+ }
+ /*
+ * The alternate request queue that we may end up submitting
+ * the bio to may be frozen temporarily, in this case REQ_NOWAIT
+ * will fail the I/O immediately with EAGAIN to the issuer.
+ * We are not in the issuer context which cannot block. Clear
+ * the flag to avoid spurious EAGAIN I/O failures.
+ */
+ bio->bi_opf &= ~REQ_NOWAIT;
+ bio_clear_flag(bio, BIO_QOS_THROTTLED);
+ bio_clear_flag(bio, BIO_QOS_MERGED);
+ }
+
if (rq->bio) {
if (list->tail)
list->tail->bi_next = rq->bio;
diff --git a/block/blk-settings.c b/block/blk-settings.c
index dabfab97fbab..78c83817b9d3 100644
--- a/block/blk-settings.c
+++ b/block/blk-settings.c
@@ -189,11 +189,11 @@ static int blk_validate_integrity_limits(struct queue_limits *lim)
}
/*
- * The PI generation / validation helpers do not expect intervals to
- * straddle multiple bio_vecs. Enforce alignment so that those are
+ * Some IO controllers can not handle data intervals straddling
+ * multiple bio_vecs. For those, enforce alignment so that those are
* never generated, and that each buffer is aligned as expected.
*/
- if (bi->csum_type) {
+ if (!(bi->flags & BLK_SPLIT_INTERVAL_CAPABLE) && bi->csum_type) {
lim->dma_alignment = max(lim->dma_alignment,
(1U << bi->interval_exp) - 1);
}
@@ -992,10 +992,14 @@ bool queue_limits_stack_integrity(struct queue_limits *t,
if ((ti->flags & BLK_INTEGRITY_REF_TAG) !=
(bi->flags & BLK_INTEGRITY_REF_TAG))
goto incompatible;
+ if ((ti->flags & BLK_SPLIT_INTERVAL_CAPABLE) &&
+ !(bi->flags & BLK_SPLIT_INTERVAL_CAPABLE))
+ ti->flags &= ~BLK_SPLIT_INTERVAL_CAPABLE;
} else {
ti->flags = BLK_INTEGRITY_STACKED;
ti->flags |= (bi->flags & BLK_INTEGRITY_DEVICE_CAPABLE) |
- (bi->flags & BLK_INTEGRITY_REF_TAG);
+ (bi->flags & BLK_INTEGRITY_REF_TAG) |
+ (bi->flags & BLK_SPLIT_INTERVAL_CAPABLE);
ti->csum_type = bi->csum_type;
ti->pi_tuple_size = bi->pi_tuple_size;
ti->metadata_size = bi->metadata_size;
diff --git a/block/blk-sysfs.c b/block/blk-sysfs.c
index 55a1bbfef7d4..f22c1f253eb3 100644
--- a/block/blk-sysfs.c
+++ b/block/blk-sysfs.c
@@ -390,6 +390,36 @@ static ssize_t queue_nr_zones_show(struct gendisk *disk, char *page)
return queue_var_show(disk_nr_zones(disk), page);
}
+static ssize_t queue_zoned_qd1_writes_show(struct gendisk *disk, char *page)
+{
+ return queue_var_show(!!blk_queue_zoned_qd1_writes(disk->queue),
+ page);
+}
+
+static ssize_t queue_zoned_qd1_writes_store(struct gendisk *disk,
+ const char *page, size_t count)
+{
+ struct request_queue *q = disk->queue;
+ unsigned long qd1_writes;
+ unsigned int memflags;
+ ssize_t ret;
+
+ ret = queue_var_store(&qd1_writes, page, count);
+ if (ret < 0)
+ return ret;
+
+ memflags = blk_mq_freeze_queue(q);
+ blk_mq_quiesce_queue(q);
+ if (qd1_writes)
+ blk_queue_flag_set(QUEUE_FLAG_ZONED_QD1_WRITES, q);
+ else
+ blk_queue_flag_clear(QUEUE_FLAG_ZONED_QD1_WRITES, q);
+ blk_mq_unquiesce_queue(q);
+ blk_mq_unfreeze_queue(q, memflags);
+
+ return count;
+}
+
static ssize_t queue_iostats_passthrough_show(struct gendisk *disk, char *page)
{
return queue_var_show(!!blk_queue_passthrough_stat(disk->queue), page);
@@ -551,27 +581,27 @@ static int queue_wc_store(struct gendisk *disk, const char *page,
return 0;
}
-#define QUEUE_RO_ENTRY(_prefix, _name) \
-static struct queue_sysfs_entry _prefix##_entry = { \
- .attr = { .name = _name, .mode = 0444 }, \
- .show = _prefix##_show, \
+#define QUEUE_RO_ENTRY(_prefix, _name) \
+static const struct queue_sysfs_entry _prefix##_entry = { \
+ .attr = { .name = _name, .mode = 0444 }, \
+ .show = _prefix##_show, \
};
-#define QUEUE_RW_ENTRY(_prefix, _name) \
-static struct queue_sysfs_entry _prefix##_entry = { \
- .attr = { .name = _name, .mode = 0644 }, \
- .show = _prefix##_show, \
- .store = _prefix##_store, \
+#define QUEUE_RW_ENTRY(_prefix, _name) \
+static const struct queue_sysfs_entry _prefix##_entry = { \
+ .attr = { .name = _name, .mode = 0644 }, \
+ .show = _prefix##_show, \
+ .store = _prefix##_store, \
};
#define QUEUE_LIM_RO_ENTRY(_prefix, _name) \
-static struct queue_sysfs_entry _prefix##_entry = { \
+static const struct queue_sysfs_entry _prefix##_entry = { \
.attr = { .name = _name, .mode = 0444 }, \
.show_limit = _prefix##_show, \
}
#define QUEUE_LIM_RW_ENTRY(_prefix, _name) \
-static struct queue_sysfs_entry _prefix##_entry = { \
+static const struct queue_sysfs_entry _prefix##_entry = { \
.attr = { .name = _name, .mode = 0644 }, \
.show_limit = _prefix##_show, \
.store_limit = _prefix##_store, \
@@ -617,6 +647,7 @@ QUEUE_LIM_RO_ENTRY(queue_max_zone_append_sectors, "zone_append_max_bytes");
QUEUE_LIM_RO_ENTRY(queue_zone_write_granularity, "zone_write_granularity");
QUEUE_LIM_RO_ENTRY(queue_zoned, "zoned");
+QUEUE_RW_ENTRY(queue_zoned_qd1_writes, "zoned_qd1_writes");
QUEUE_RO_ENTRY(queue_nr_zones, "nr_zones");
QUEUE_LIM_RO_ENTRY(queue_max_open_zones, "max_open_zones");
QUEUE_LIM_RO_ENTRY(queue_max_active_zones, "max_active_zones");
@@ -634,7 +665,7 @@ QUEUE_LIM_RO_ENTRY(queue_virt_boundary_mask, "virt_boundary_mask");
QUEUE_LIM_RO_ENTRY(queue_dma_alignment, "dma_alignment");
/* legacy alias for logical_block_size: */
-static struct queue_sysfs_entry queue_hw_sector_size_entry = {
+static const struct queue_sysfs_entry queue_hw_sector_size_entry = {
.attr = {.name = "hw_sector_size", .mode = 0444 },
.show_limit = queue_logical_block_size_show,
};
@@ -700,7 +731,7 @@ QUEUE_RW_ENTRY(queue_wb_lat, "wbt_lat_usec");
#endif
/* Common attributes for bio-based and request-based queues. */
-static struct attribute *queue_attrs[] = {
+static const struct attribute *const queue_attrs[] = {
/*
* Attributes which are protected with q->limits_lock.
*/
@@ -754,12 +785,13 @@ static struct attribute *queue_attrs[] = {
&queue_nomerges_entry.attr,
&queue_poll_entry.attr,
&queue_poll_delay_entry.attr,
+ &queue_zoned_qd1_writes_entry.attr,
NULL,
};
/* Request-based queue attributes that are not relevant for bio-based queues. */
-static struct attribute *blk_mq_queue_attrs[] = {
+static const struct attribute *const blk_mq_queue_attrs[] = {
/*
* Attributes which require some form of locking other than
* q->sysfs_lock.
@@ -779,14 +811,15 @@ static struct attribute *blk_mq_queue_attrs[] = {
NULL,
};
-static umode_t queue_attr_visible(struct kobject *kobj, struct attribute *attr,
+static umode_t queue_attr_visible(struct kobject *kobj, const struct attribute *attr,
int n)
{
struct gendisk *disk = container_of(kobj, struct gendisk, queue_kobj);
struct request_queue *q = disk->queue;
if ((attr == &queue_max_open_zones_entry.attr ||
- attr == &queue_max_active_zones_entry.attr) &&
+ attr == &queue_max_active_zones_entry.attr ||
+ attr == &queue_zoned_qd1_writes_entry.attr) &&
!blk_queue_is_zoned(q))
return 0;
@@ -794,7 +827,7 @@ static umode_t queue_attr_visible(struct kobject *kobj, struct attribute *attr,
}
static umode_t blk_mq_queue_attr_visible(struct kobject *kobj,
- struct attribute *attr, int n)
+ const struct attribute *attr, int n)
{
struct gendisk *disk = container_of(kobj, struct gendisk, queue_kobj);
struct request_queue *q = disk->queue;
@@ -808,17 +841,17 @@ static umode_t blk_mq_queue_attr_visible(struct kobject *kobj,
return attr->mode;
}
-static struct attribute_group queue_attr_group = {
- .attrs = queue_attrs,
- .is_visible = queue_attr_visible,
+static const struct attribute_group queue_attr_group = {
+ .attrs_const = queue_attrs,
+ .is_visible_const = queue_attr_visible,
};
-static struct attribute_group blk_mq_queue_attr_group = {
- .attrs = blk_mq_queue_attrs,
- .is_visible = blk_mq_queue_attr_visible,
+static const struct attribute_group blk_mq_queue_attr_group = {
+ .attrs_const = blk_mq_queue_attrs,
+ .is_visible_const = blk_mq_queue_attr_visible,
};
-#define to_queue(atr) container_of((atr), struct queue_sysfs_entry, attr)
+#define to_queue(atr) container_of_const((atr), struct queue_sysfs_entry, attr)
static ssize_t
queue_attr_show(struct kobject *kobj, struct attribute *attr, char *page)
@@ -934,6 +967,14 @@ int blk_register_queue(struct gendisk *disk)
blk_mq_debugfs_register(q);
blk_debugfs_unlock(q, memflags);
+ /*
+ * For blk-mq rotational zoned devices, default to using QD=1
+ * writes. For non-mq rotational zoned devices, the device driver can
+ * set an appropriate default.
+ */
+ if (queue_is_mq(q) && blk_queue_rot(q) && blk_queue_is_zoned(q))
+ blk_queue_flag_set(QUEUE_FLAG_ZONED_QD1_WRITES, q);
+
ret = disk_register_independent_access_ranges(disk);
if (ret)
goto out_debugfs_remove;
diff --git a/block/blk-wbt.c b/block/blk-wbt.c
index 33006edfccd4..dcc2438ca16d 100644
--- a/block/blk-wbt.c
+++ b/block/blk-wbt.c
@@ -782,10 +782,11 @@ void wbt_init_enable_default(struct gendisk *disk)
return;
rwb = wbt_alloc();
- if (WARN_ON_ONCE(!rwb))
+ if (!rwb)
return;
- if (WARN_ON_ONCE(wbt_init(disk, rwb))) {
+ if (wbt_init(disk, rwb)) {
+ pr_warn("%s: failed to enable wbt\n", disk->disk_name);
wbt_free(rwb);
return;
}
diff --git a/block/blk-zoned.c b/block/blk-zoned.c
index 9d1dd6ccfad7..30cad2bb9291 100644
--- a/block/blk-zoned.c
+++ b/block/blk-zoned.c
@@ -16,6 +16,8 @@
#include <linux/spinlock.h>
#include <linux/refcount.h>
#include <linux/mempool.h>
+#include <linux/kthread.h>
+#include <linux/freezer.h>
#include <trace/events/block.h>
@@ -40,6 +42,8 @@ static const char *const zone_cond_name[] = {
/*
* Per-zone write plug.
* @node: hlist_node structure for managing the plug using a hash table.
+ * @entry: list_head structure for listing the plug in the disk list of active
+ * zone write plugs.
* @bio_list: The list of BIOs that are currently plugged.
* @bio_work: Work struct to handle issuing of plugged BIOs
* @rcu_head: RCU head to free zone write plugs with an RCU grace period.
@@ -62,6 +66,7 @@ static const char *const zone_cond_name[] = {
*/
struct blk_zone_wplug {
struct hlist_node node;
+ struct list_head entry;
struct bio_list bio_list;
struct work_struct bio_work;
struct rcu_head rcu_head;
@@ -99,17 +104,17 @@ static inline unsigned int disk_zone_wplugs_hash_size(struct gendisk *disk)
* being executed or the zone write plug bio list is not empty.
* - BLK_ZONE_WPLUG_NEED_WP_UPDATE: Indicates that we lost track of a zone
* write pointer offset and need to update it.
- * - BLK_ZONE_WPLUG_UNHASHED: Indicates that the zone write plug was removed
- * from the disk hash table and that the initial reference to the zone
- * write plug set when the plug was first added to the hash table has been
- * dropped. This flag is set when a zone is reset, finished or become full,
- * to prevent new references to the zone write plug to be taken for
- * newly incoming BIOs. A zone write plug flagged with this flag will be
- * freed once all remaining references from BIOs or functions are dropped.
+ * - BLK_ZONE_WPLUG_DEAD: Indicates that the zone write plug will be
+ * removed from the disk hash table of zone write plugs when the last
+ * reference on the zone write plug is dropped. If set, this flag also
+ * indicates that the initial extra reference on the zone write plug was
+ * dropped, meaning that the reference count indicates the current number of
+ * active users (code context or BIOs and requests in flight). This flag is
+ * set when a zone is reset, finished or becomes full.
*/
#define BLK_ZONE_WPLUG_PLUGGED (1U << 0)
#define BLK_ZONE_WPLUG_NEED_WP_UPDATE (1U << 1)
-#define BLK_ZONE_WPLUG_UNHASHED (1U << 2)
+#define BLK_ZONE_WPLUG_DEAD (1U << 2)
/**
* blk_zone_cond_str - Return a zone condition name string
@@ -412,20 +417,32 @@ int blkdev_report_zones_ioctl(struct block_device *bdev, unsigned int cmd,
return 0;
}
-static int blkdev_truncate_zone_range(struct block_device *bdev,
- blk_mode_t mode, const struct blk_zone_range *zrange)
+static int blkdev_reset_zone(struct block_device *bdev, blk_mode_t mode,
+ struct blk_zone_range *zrange)
{
loff_t start, end;
+ int ret = -EINVAL;
+ inode_lock(bdev->bd_mapping->host);
+ filemap_invalidate_lock(bdev->bd_mapping);
if (zrange->sector + zrange->nr_sectors <= zrange->sector ||
zrange->sector + zrange->nr_sectors > get_capacity(bdev->bd_disk))
/* Out of range */
- return -EINVAL;
+ goto out_unlock;
start = zrange->sector << SECTOR_SHIFT;
end = ((zrange->sector + zrange->nr_sectors) << SECTOR_SHIFT) - 1;
- return truncate_bdev_range(bdev, mode, start, end);
+ ret = truncate_bdev_range(bdev, mode, start, end);
+ if (ret)
+ goto out_unlock;
+
+ ret = blkdev_zone_mgmt(bdev, REQ_OP_ZONE_RESET, zrange->sector,
+ zrange->nr_sectors);
+out_unlock:
+ filemap_invalidate_unlock(bdev->bd_mapping);
+ inode_unlock(bdev->bd_mapping->host);
+ return ret;
}
/*
@@ -438,7 +455,6 @@ int blkdev_zone_mgmt_ioctl(struct block_device *bdev, blk_mode_t mode,
void __user *argp = (void __user *)arg;
struct blk_zone_range zrange;
enum req_op op;
- int ret;
if (!argp)
return -EINVAL;
@@ -454,15 +470,7 @@ int blkdev_zone_mgmt_ioctl(struct block_device *bdev, blk_mode_t mode,
switch (cmd) {
case BLKRESETZONE:
- op = REQ_OP_ZONE_RESET;
-
- /* Invalidate the page cache, including dirty pages. */
- inode_lock(bdev->bd_mapping->host);
- filemap_invalidate_lock(bdev->bd_mapping);
- ret = blkdev_truncate_zone_range(bdev, mode, &zrange);
- if (ret)
- goto fail;
- break;
+ return blkdev_reset_zone(bdev, mode, &zrange);
case BLKOPENZONE:
op = REQ_OP_ZONE_OPEN;
break;
@@ -476,15 +484,7 @@ int blkdev_zone_mgmt_ioctl(struct block_device *bdev, blk_mode_t mode,
return -ENOTTY;
}
- ret = blkdev_zone_mgmt(bdev, op, zrange.sector, zrange.nr_sectors);
-
-fail:
- if (cmd == BLKRESETZONE) {
- filemap_invalidate_unlock(bdev->bd_mapping);
- inode_unlock(bdev->bd_mapping->host);
- }
-
- return ret;
+ return blkdev_zone_mgmt(bdev, op, zrange.sector, zrange.nr_sectors);
}
static bool disk_zone_is_last(struct gendisk *disk, struct blk_zone *zone)
@@ -492,18 +492,12 @@ static bool disk_zone_is_last(struct gendisk *disk, struct blk_zone *zone)
return zone->start + zone->len >= get_capacity(disk);
}
-static bool disk_zone_is_full(struct gendisk *disk,
- unsigned int zno, unsigned int offset_in_zone)
-{
- if (zno < disk->nr_zones - 1)
- return offset_in_zone >= disk->zone_capacity;
- return offset_in_zone >= disk->last_zone_capacity;
-}
-
static bool disk_zone_wplug_is_full(struct gendisk *disk,
struct blk_zone_wplug *zwplug)
{
- return disk_zone_is_full(disk, zwplug->zone_no, zwplug->wp_offset);
+ if (zwplug->zone_no < disk->nr_zones - 1)
+ return zwplug->wp_offset >= disk->zone_capacity;
+ return zwplug->wp_offset >= disk->last_zone_capacity;
}
static bool disk_insert_zone_wplug(struct gendisk *disk,
@@ -520,10 +514,11 @@ static bool disk_insert_zone_wplug(struct gendisk *disk,
* are racing with other submission context, so we may already have a
* zone write plug for the same zone.
*/
- spin_lock_irqsave(&disk->zone_wplugs_lock, flags);
+ spin_lock_irqsave(&disk->zone_wplugs_hash_lock, flags);
hlist_for_each_entry_rcu(zwplg, &disk->zone_wplugs_hash[idx], node) {
if (zwplg->zone_no == zwplug->zone_no) {
- spin_unlock_irqrestore(&disk->zone_wplugs_lock, flags);
+ spin_unlock_irqrestore(&disk->zone_wplugs_hash_lock,
+ flags);
return false;
}
}
@@ -535,7 +530,7 @@ static bool disk_insert_zone_wplug(struct gendisk *disk,
* necessarilly in the active condition.
*/
zones_cond = rcu_dereference_check(disk->zones_cond,
- lockdep_is_held(&disk->zone_wplugs_lock));
+ lockdep_is_held(&disk->zone_wplugs_hash_lock));
if (zones_cond)
zwplug->cond = zones_cond[zwplug->zone_no];
else
@@ -543,7 +538,7 @@ static bool disk_insert_zone_wplug(struct gendisk *disk,
hlist_add_head_rcu(&zwplug->node, &disk->zone_wplugs_hash[idx]);
atomic_inc(&disk->nr_zone_wplugs);
- spin_unlock_irqrestore(&disk->zone_wplugs_lock, flags);
+ spin_unlock_irqrestore(&disk->zone_wplugs_hash_lock, flags);
return true;
}
@@ -587,105 +582,76 @@ static void disk_free_zone_wplug_rcu(struct rcu_head *rcu_head)
mempool_free(zwplug, zwplug->disk->zone_wplugs_pool);
}
-static inline void disk_put_zone_wplug(struct blk_zone_wplug *zwplug)
+static void disk_free_zone_wplug(struct blk_zone_wplug *zwplug)
{
- if (refcount_dec_and_test(&zwplug->ref)) {
- WARN_ON_ONCE(!bio_list_empty(&zwplug->bio_list));
- WARN_ON_ONCE(zwplug->flags & BLK_ZONE_WPLUG_PLUGGED);
- WARN_ON_ONCE(!(zwplug->flags & BLK_ZONE_WPLUG_UNHASHED));
+ struct gendisk *disk = zwplug->disk;
+ unsigned long flags;
- call_rcu(&zwplug->rcu_head, disk_free_zone_wplug_rcu);
- }
-}
+ WARN_ON_ONCE(!(zwplug->flags & BLK_ZONE_WPLUG_DEAD));
+ WARN_ON_ONCE(zwplug->flags & BLK_ZONE_WPLUG_PLUGGED);
+ WARN_ON_ONCE(!bio_list_empty(&zwplug->bio_list));
-static inline bool disk_should_remove_zone_wplug(struct gendisk *disk,
- struct blk_zone_wplug *zwplug)
-{
- lockdep_assert_held(&zwplug->lock);
+ spin_lock_irqsave(&disk->zone_wplugs_hash_lock, flags);
+ blk_zone_set_cond(rcu_dereference_check(disk->zones_cond,
+ lockdep_is_held(&disk->zone_wplugs_hash_lock)),
+ zwplug->zone_no, zwplug->cond);
+ hlist_del_init_rcu(&zwplug->node);
+ atomic_dec(&disk->nr_zone_wplugs);
+ spin_unlock_irqrestore(&disk->zone_wplugs_hash_lock, flags);
- /* If the zone write plug was already removed, we are done. */
- if (zwplug->flags & BLK_ZONE_WPLUG_UNHASHED)
- return false;
+ call_rcu(&zwplug->rcu_head, disk_free_zone_wplug_rcu);
+}
- /* If the zone write plug is still plugged, it cannot be removed. */
- if (zwplug->flags & BLK_ZONE_WPLUG_PLUGGED)
- return false;
+static inline void disk_put_zone_wplug(struct blk_zone_wplug *zwplug)
+{
+ if (refcount_dec_and_test(&zwplug->ref))
+ disk_free_zone_wplug(zwplug);
+}
- /*
- * Completions of BIOs with blk_zone_write_plug_bio_endio() may
- * happen after handling a request completion with
- * blk_zone_write_plug_finish_request() (e.g. with split BIOs
- * that are chained). In such case, disk_zone_wplug_unplug_bio()
- * should not attempt to remove the zone write plug until all BIO
- * completions are seen. Check by looking at the zone write plug
- * reference count, which is 2 when the plug is unused (one reference
- * taken when the plug was allocated and another reference taken by the
- * caller context).
- */
- if (refcount_read(&zwplug->ref) > 2)
- return false;
+/*
+ * Flag the zone write plug as dead and drop the initial reference we got when
+ * the zone write plug was added to the hash table. The zone write plug will be
+ * unhashed when its last reference is dropped.
+ */
+static void disk_mark_zone_wplug_dead(struct blk_zone_wplug *zwplug)
+{
+ lockdep_assert_held(&zwplug->lock);
- /* We can remove zone write plugs for zones that are empty or full. */
- return !zwplug->wp_offset || disk_zone_wplug_is_full(disk, zwplug);
+ if (!(zwplug->flags & BLK_ZONE_WPLUG_DEAD)) {
+ zwplug->flags |= BLK_ZONE_WPLUG_DEAD;
+ disk_put_zone_wplug(zwplug);
+ }
}
-static void disk_remove_zone_wplug(struct gendisk *disk,
- struct blk_zone_wplug *zwplug)
+static bool disk_zone_wplug_submit_bio(struct gendisk *disk,
+ struct blk_zone_wplug *zwplug);
+
+static void blk_zone_wplug_bio_work(struct work_struct *work)
{
- unsigned long flags;
+ struct blk_zone_wplug *zwplug =
+ container_of(work, struct blk_zone_wplug, bio_work);
- /* If the zone write plug was already removed, we have nothing to do. */
- if (zwplug->flags & BLK_ZONE_WPLUG_UNHASHED)
- return;
+ disk_zone_wplug_submit_bio(zwplug->disk, zwplug);
- /*
- * Mark the zone write plug as unhashed and drop the extra reference we
- * took when the plug was inserted in the hash table. Also update the
- * disk zone condition array with the current condition of the zone
- * write plug.
- */
- zwplug->flags |= BLK_ZONE_WPLUG_UNHASHED;
- spin_lock_irqsave(&disk->zone_wplugs_lock, flags);
- blk_zone_set_cond(rcu_dereference_check(disk->zones_cond,
- lockdep_is_held(&disk->zone_wplugs_lock)),
- zwplug->zone_no, zwplug->cond);
- hlist_del_init_rcu(&zwplug->node);
- atomic_dec(&disk->nr_zone_wplugs);
- spin_unlock_irqrestore(&disk->zone_wplugs_lock, flags);
+ /* Drop the reference we took in disk_zone_wplug_schedule_work(). */
disk_put_zone_wplug(zwplug);
}
-static void blk_zone_wplug_bio_work(struct work_struct *work);
-
/*
- * Get a reference on the write plug for the zone containing @sector.
- * If the plug does not exist, it is allocated and hashed.
- * Return a pointer to the zone write plug with the plug spinlock held.
+ * Get a zone write plug for the zone containing @sector.
+ * If the plug does not exist, it is allocated and inserted in the disk hash
+ * table.
*/
-static struct blk_zone_wplug *disk_get_and_lock_zone_wplug(struct gendisk *disk,
- sector_t sector, gfp_t gfp_mask,
- unsigned long *flags)
+static struct blk_zone_wplug *disk_get_or_alloc_zone_wplug(struct gendisk *disk,
+ sector_t sector, gfp_t gfp_mask)
{
unsigned int zno = disk_zone_no(disk, sector);
struct blk_zone_wplug *zwplug;
again:
zwplug = disk_get_zone_wplug(disk, sector);
- if (zwplug) {
- /*
- * Check that a BIO completion or a zone reset or finish
- * operation has not already removed the zone write plug from
- * the hash table and dropped its reference count. In such case,
- * we need to get a new plug so start over from the beginning.
- */
- spin_lock_irqsave(&zwplug->lock, *flags);
- if (zwplug->flags & BLK_ZONE_WPLUG_UNHASHED) {
- spin_unlock_irqrestore(&zwplug->lock, *flags);
- disk_put_zone_wplug(zwplug);
- goto again;
- }
+ if (zwplug)
return zwplug;
- }
/*
* Allocate and initialize a zone write plug with an extra reference
@@ -704,17 +670,15 @@ again:
zwplug->wp_offset = bdev_offset_from_zone_start(disk->part0, sector);
bio_list_init(&zwplug->bio_list);
INIT_WORK(&zwplug->bio_work, blk_zone_wplug_bio_work);
+ INIT_LIST_HEAD(&zwplug->entry);
zwplug->disk = disk;
- spin_lock_irqsave(&zwplug->lock, *flags);
-
/*
* Insert the new zone write plug in the hash table. This can fail only
* if another context already inserted a plug. Retry from the beginning
* in such case.
*/
if (!disk_insert_zone_wplug(disk, zwplug)) {
- spin_unlock_irqrestore(&zwplug->lock, *flags);
mempool_free(zwplug, disk->zone_wplugs_pool);
goto again;
}
@@ -739,6 +703,7 @@ static inline void blk_zone_wplug_bio_io_error(struct blk_zone_wplug *zwplug,
*/
static void disk_zone_wplug_abort(struct blk_zone_wplug *zwplug)
{
+ struct gendisk *disk = zwplug->disk;
struct bio *bio;
lockdep_assert_held(&zwplug->lock);
@@ -752,6 +717,20 @@ static void disk_zone_wplug_abort(struct blk_zone_wplug *zwplug)
blk_zone_wplug_bio_io_error(zwplug, bio);
zwplug->flags &= ~BLK_ZONE_WPLUG_PLUGGED;
+
+ /*
+ * If we are using the per disk zone write plugs worker thread, remove
+ * the zone write plug from the work list and drop the reference we
+ * took when the zone write plug was added to that list.
+ */
+ if (blk_queue_zoned_qd1_writes(disk->queue)) {
+ spin_lock(&disk->zone_wplugs_list_lock);
+ if (!list_empty(&zwplug->entry)) {
+ list_del_init(&zwplug->entry);
+ disk_put_zone_wplug(zwplug);
+ }
+ spin_unlock(&disk->zone_wplugs_list_lock);
+ }
}
/*
@@ -788,14 +767,8 @@ static void disk_zone_wplug_set_wp_offset(struct gendisk *disk,
disk_zone_wplug_update_cond(disk, zwplug);
disk_zone_wplug_abort(zwplug);
-
- /*
- * The zone write plug now has no BIO plugged: remove it from the
- * hash table so that it cannot be seen. The plug will be freed
- * when the last reference is dropped.
- */
- if (disk_should_remove_zone_wplug(disk, zwplug))
- disk_remove_zone_wplug(disk, zwplug);
+ if (!zwplug->wp_offset || disk_zone_wplug_is_full(disk, zwplug))
+ disk_mark_zone_wplug_dead(zwplug);
}
static unsigned int blk_zone_wp_offset(struct blk_zone *zone)
@@ -1192,19 +1165,24 @@ void blk_zone_mgmt_bio_endio(struct bio *bio)
}
}
-static void disk_zone_wplug_schedule_bio_work(struct gendisk *disk,
- struct blk_zone_wplug *zwplug)
+static void disk_zone_wplug_schedule_work(struct gendisk *disk,
+ struct blk_zone_wplug *zwplug)
{
lockdep_assert_held(&zwplug->lock);
/*
- * Take a reference on the zone write plug and schedule the submission
- * of the next plugged BIO. blk_zone_wplug_bio_work() will release the
- * reference we take here.
+ * Schedule the submission of the next plugged BIO. Taking a reference
+ * to the zone write plug is required as the bio_work belongs to the
+ * plug, and thus we must ensure that the write plug does not go away
+ * while the work is being scheduled but has not run yet.
+ * blk_zone_wplug_bio_work() will release the reference we take here,
+ * and we also drop this reference if the work is already scheduled.
*/
WARN_ON_ONCE(!(zwplug->flags & BLK_ZONE_WPLUG_PLUGGED));
+ WARN_ON_ONCE(blk_queue_zoned_qd1_writes(disk->queue));
refcount_inc(&zwplug->ref);
- queue_work(disk->zone_wplugs_wq, &zwplug->bio_work);
+ if (!queue_work(disk->zone_wplugs_wq, &zwplug->bio_work))
+ disk_put_zone_wplug(zwplug);
}
static inline void disk_zone_wplug_add_bio(struct gendisk *disk,
@@ -1241,6 +1219,22 @@ static inline void disk_zone_wplug_add_bio(struct gendisk *disk,
bio_list_add(&zwplug->bio_list, bio);
trace_disk_zone_wplug_add_bio(zwplug->disk->queue, zwplug->zone_no,
bio->bi_iter.bi_sector, bio_sectors(bio));
+
+ /*
+ * If we are using the disk zone write plugs worker instead of the per
+ * zone write plug BIO work, add the zone write plug to the work list
+ * if it is not already there. Make sure to also get an extra reference
+ * on the zone write plug so that it does not go away until it is
+ * removed from the work list.
+ */
+ if (blk_queue_zoned_qd1_writes(disk->queue)) {
+ spin_lock(&disk->zone_wplugs_list_lock);
+ if (list_empty(&zwplug->entry)) {
+ list_add_tail(&zwplug->entry, &disk->zone_wplugs_list);
+ refcount_inc(&zwplug->ref);
+ }
+ spin_unlock(&disk->zone_wplugs_list_lock);
+ }
}
/*
@@ -1438,7 +1432,7 @@ static bool blk_zone_wplug_handle_write(struct bio *bio, unsigned int nr_segs)
if (bio->bi_opf & REQ_NOWAIT)
gfp_mask = GFP_NOWAIT;
- zwplug = disk_get_and_lock_zone_wplug(disk, sector, gfp_mask, &flags);
+ zwplug = disk_get_or_alloc_zone_wplug(disk, sector, gfp_mask);
if (!zwplug) {
if (bio->bi_opf & REQ_NOWAIT)
bio_wouldblock_error(bio);
@@ -1447,6 +1441,21 @@ static bool blk_zone_wplug_handle_write(struct bio *bio, unsigned int nr_segs)
return true;
}
+ spin_lock_irqsave(&zwplug->lock, flags);
+
+ /*
+ * If we got a zone write plug marked as dead, then the user is issuing
+ * writes to a full zone, or without synchronizing with zone reset or
+ * zone finish operations. In such case, fail the BIO to signal this
+ * invalid usage.
+ */
+ if (zwplug->flags & BLK_ZONE_WPLUG_DEAD) {
+ spin_unlock_irqrestore(&zwplug->lock, flags);
+ disk_put_zone_wplug(zwplug);
+ bio_io_error(bio);
+ return true;
+ }
+
/* Indicate that this BIO is being handled using zone write plugging. */
bio_set_flag(bio, BIO_ZONE_WRITE_PLUGGING);
@@ -1459,6 +1468,13 @@ static bool blk_zone_wplug_handle_write(struct bio *bio, unsigned int nr_segs)
goto queue_bio;
}
+ /*
+ * For rotational devices, we will use the gendisk zone write plugs
+ * work instead of the per zone write plug BIO work, so queue the BIO.
+ */
+ if (blk_queue_zoned_qd1_writes(disk->queue))
+ goto queue_bio;
+
/* If the zone is already plugged, add the BIO to the BIO plug list. */
if (zwplug->flags & BLK_ZONE_WPLUG_PLUGGED)
goto queue_bio;
@@ -1481,7 +1497,10 @@ queue_bio:
if (!(zwplug->flags & BLK_ZONE_WPLUG_PLUGGED)) {
zwplug->flags |= BLK_ZONE_WPLUG_PLUGGED;
- disk_zone_wplug_schedule_bio_work(disk, zwplug);
+ if (blk_queue_zoned_qd1_writes(disk->queue))
+ wake_up_process(disk->zone_wplugs_worker);
+ else
+ disk_zone_wplug_schedule_work(disk, zwplug);
}
spin_unlock_irqrestore(&zwplug->lock, flags);
@@ -1527,7 +1546,7 @@ static void blk_zone_wplug_handle_native_zone_append(struct bio *bio)
disk->disk_name, zwplug->zone_no);
disk_zone_wplug_abort(zwplug);
}
- disk_remove_zone_wplug(disk, zwplug);
+ disk_mark_zone_wplug_dead(zwplug);
spin_unlock_irqrestore(&zwplug->lock, flags);
disk_put_zone_wplug(zwplug);
@@ -1622,21 +1641,21 @@ static void disk_zone_wplug_unplug_bio(struct gendisk *disk,
spin_lock_irqsave(&zwplug->lock, flags);
- /* Schedule submission of the next plugged BIO if we have one. */
- if (!bio_list_empty(&zwplug->bio_list)) {
- disk_zone_wplug_schedule_bio_work(disk, zwplug);
- spin_unlock_irqrestore(&zwplug->lock, flags);
- return;
- }
-
- zwplug->flags &= ~BLK_ZONE_WPLUG_PLUGGED;
-
/*
- * If the zone is full (it was fully written or finished, or empty
- * (it was reset), remove its zone write plug from the hash table.
+ * For rotational devices, signal the BIO completion to the zone write
+ * plug work. Otherwise, schedule submission of the next plugged BIO
+ * if we have one.
*/
- if (disk_should_remove_zone_wplug(disk, zwplug))
- disk_remove_zone_wplug(disk, zwplug);
+ if (bio_list_empty(&zwplug->bio_list))
+ zwplug->flags &= ~BLK_ZONE_WPLUG_PLUGGED;
+
+ if (blk_queue_zoned_qd1_writes(disk->queue))
+ complete(&disk->zone_wplugs_worker_bio_done);
+ else if (!bio_list_empty(&zwplug->bio_list))
+ disk_zone_wplug_schedule_work(disk, zwplug);
+
+ if (!zwplug->wp_offset || disk_zone_wplug_is_full(disk, zwplug))
+ disk_mark_zone_wplug_dead(zwplug);
spin_unlock_irqrestore(&zwplug->lock, flags);
}
@@ -1727,10 +1746,9 @@ void blk_zone_write_plug_finish_request(struct request *req)
disk_put_zone_wplug(zwplug);
}
-static void blk_zone_wplug_bio_work(struct work_struct *work)
+static bool disk_zone_wplug_submit_bio(struct gendisk *disk,
+ struct blk_zone_wplug *zwplug)
{
- struct blk_zone_wplug *zwplug =
- container_of(work, struct blk_zone_wplug, bio_work);
struct block_device *bdev;
unsigned long flags;
struct bio *bio;
@@ -1746,7 +1764,7 @@ again:
if (!bio) {
zwplug->flags &= ~BLK_ZONE_WPLUG_PLUGGED;
spin_unlock_irqrestore(&zwplug->lock, flags);
- goto put_zwplug;
+ return false;
}
trace_blk_zone_wplug_bio(zwplug->disk->queue, zwplug->zone_no,
@@ -1760,14 +1778,15 @@ again:
goto again;
}
- bdev = bio->bi_bdev;
-
/*
* blk-mq devices will reuse the extra reference on the request queue
* usage counter we took when the BIO was plugged, but the submission
* path for BIO-based devices will not do that. So drop this extra
* reference here.
*/
+ if (blk_queue_zoned_qd1_writes(disk->queue))
+ reinit_completion(&disk->zone_wplugs_worker_bio_done);
+ bdev = bio->bi_bdev;
if (bdev_test_flag(bdev, BD_HAS_SUBMIT_BIO)) {
bdev->bd_disk->fops->submit_bio(bio);
blk_queue_exit(bdev->bd_disk->queue);
@@ -1775,14 +1794,78 @@ again:
blk_mq_submit_bio(bio);
}
-put_zwplug:
- /* Drop the reference we took in disk_zone_wplug_schedule_bio_work(). */
- disk_put_zone_wplug(zwplug);
+ return true;
+}
+
+static struct blk_zone_wplug *disk_get_zone_wplugs_work(struct gendisk *disk)
+{
+ struct blk_zone_wplug *zwplug;
+
+ spin_lock_irq(&disk->zone_wplugs_list_lock);
+ zwplug = list_first_entry_or_null(&disk->zone_wplugs_list,
+ struct blk_zone_wplug, entry);
+ if (zwplug)
+ list_del_init(&zwplug->entry);
+ spin_unlock_irq(&disk->zone_wplugs_list_lock);
+
+ return zwplug;
+}
+
+static int disk_zone_wplugs_worker(void *data)
+{
+ struct gendisk *disk = data;
+ struct blk_zone_wplug *zwplug;
+ unsigned int noio_flag;
+
+ noio_flag = memalloc_noio_save();
+ set_user_nice(current, MIN_NICE);
+ set_freezable();
+
+ for (;;) {
+ set_current_state(TASK_INTERRUPTIBLE | TASK_FREEZABLE);
+
+ zwplug = disk_get_zone_wplugs_work(disk);
+ if (zwplug) {
+ /*
+ * Process all BIOs of this zone write plug and then
+ * drop the reference we took when adding the zone write
+ * plug to the active list.
+ */
+ set_current_state(TASK_RUNNING);
+ while (disk_zone_wplug_submit_bio(disk, zwplug))
+ blk_wait_io(&disk->zone_wplugs_worker_bio_done);
+ disk_put_zone_wplug(zwplug);
+ continue;
+ }
+
+ /*
+ * Only sleep if nothing sets the state to running. Else check
+ * for zone write plugs work again as a newly submitted BIO
+ * might have added a zone write plug to the work list.
+ */
+ if (get_current_state() == TASK_RUNNING) {
+ try_to_freeze();
+ } else {
+ if (kthread_should_stop()) {
+ set_current_state(TASK_RUNNING);
+ break;
+ }
+ schedule();
+ }
+ }
+
+ WARN_ON_ONCE(!list_empty(&disk->zone_wplugs_list));
+ memalloc_noio_restore(noio_flag);
+
+ return 0;
}
void disk_init_zone_resources(struct gendisk *disk)
{
- spin_lock_init(&disk->zone_wplugs_lock);
+ spin_lock_init(&disk->zone_wplugs_hash_lock);
+ spin_lock_init(&disk->zone_wplugs_list_lock);
+ INIT_LIST_HEAD(&disk->zone_wplugs_list);
+ init_completion(&disk->zone_wplugs_worker_bio_done);
}
/*
@@ -1798,6 +1881,7 @@ static int disk_alloc_zone_resources(struct gendisk *disk,
unsigned int pool_size)
{
unsigned int i;
+ int ret = -ENOMEM;
atomic_set(&disk->nr_zone_wplugs, 0);
disk->zone_wplugs_hash_bits =
@@ -1823,8 +1907,21 @@ static int disk_alloc_zone_resources(struct gendisk *disk,
if (!disk->zone_wplugs_wq)
goto destroy_pool;
+ disk->zone_wplugs_worker =
+ kthread_create(disk_zone_wplugs_worker, disk,
+ "%s_zwplugs_worker", disk->disk_name);
+ if (IS_ERR(disk->zone_wplugs_worker)) {
+ ret = PTR_ERR(disk->zone_wplugs_worker);
+ disk->zone_wplugs_worker = NULL;
+ goto destroy_wq;
+ }
+ wake_up_process(disk->zone_wplugs_worker);
+
return 0;
+destroy_wq:
+ destroy_workqueue(disk->zone_wplugs_wq);
+ disk->zone_wplugs_wq = NULL;
destroy_pool:
mempool_destroy(disk->zone_wplugs_pool);
disk->zone_wplugs_pool = NULL;
@@ -1832,7 +1929,7 @@ free_hash:
kfree(disk->zone_wplugs_hash);
disk->zone_wplugs_hash = NULL;
disk->zone_wplugs_hash_bits = 0;
- return -ENOMEM;
+ return ret;
}
static void disk_destroy_zone_wplugs_hash_table(struct gendisk *disk)
@@ -1848,9 +1945,9 @@ static void disk_destroy_zone_wplugs_hash_table(struct gendisk *disk)
while (!hlist_empty(&disk->zone_wplugs_hash[i])) {
zwplug = hlist_entry(disk->zone_wplugs_hash[i].first,
struct blk_zone_wplug, node);
- refcount_inc(&zwplug->ref);
- disk_remove_zone_wplug(disk, zwplug);
- disk_put_zone_wplug(zwplug);
+ spin_lock_irq(&zwplug->lock);
+ disk_mark_zone_wplug_dead(zwplug);
+ spin_unlock_irq(&zwplug->lock);
}
}
@@ -1872,16 +1969,20 @@ static void disk_set_zones_cond_array(struct gendisk *disk, u8 *zones_cond)
{
unsigned long flags;
- spin_lock_irqsave(&disk->zone_wplugs_lock, flags);
+ spin_lock_irqsave(&disk->zone_wplugs_hash_lock, flags);
zones_cond = rcu_replace_pointer(disk->zones_cond, zones_cond,
- lockdep_is_held(&disk->zone_wplugs_lock));
- spin_unlock_irqrestore(&disk->zone_wplugs_lock, flags);
+ lockdep_is_held(&disk->zone_wplugs_hash_lock));
+ spin_unlock_irqrestore(&disk->zone_wplugs_hash_lock, flags);
kfree_rcu_mightsleep(zones_cond);
}
void disk_free_zone_resources(struct gendisk *disk)
{
+ if (disk->zone_wplugs_worker)
+ kthread_stop(disk->zone_wplugs_worker);
+ WARN_ON_ONCE(!list_empty(&disk->zone_wplugs_list));
+
if (disk->zone_wplugs_wq) {
destroy_workqueue(disk->zone_wplugs_wq);
disk->zone_wplugs_wq = NULL;
@@ -1910,6 +2011,7 @@ static int disk_revalidate_zone_resources(struct gendisk *disk,
{
struct queue_limits *lim = &disk->queue->limits;
unsigned int pool_size;
+ int ret = 0;
args->disk = disk;
args->nr_zones =
@@ -1932,10 +2034,13 @@ static int disk_revalidate_zone_resources(struct gendisk *disk,
pool_size =
min(BLK_ZONE_WPLUG_DEFAULT_POOL_SIZE, args->nr_zones);
- if (!disk->zone_wplugs_hash)
- return disk_alloc_zone_resources(disk, pool_size);
+ if (!disk->zone_wplugs_hash) {
+ ret = disk_alloc_zone_resources(disk, pool_size);
+ if (ret)
+ kfree(args->zones_cond);
+ }
- return 0;
+ return ret;
}
/*
@@ -1967,6 +2072,7 @@ static int disk_update_zone_resources(struct gendisk *disk,
disk->zone_capacity = args->zone_capacity;
disk->last_zone_capacity = args->last_zone_capacity;
disk_set_zones_cond_array(disk, args->zones_cond);
+ args->zones_cond = NULL;
/*
* Some devices can advertise zone resource limits that are larger than
@@ -2078,7 +2184,6 @@ static int blk_revalidate_seq_zone(struct blk_zone *zone, unsigned int idx,
struct gendisk *disk = args->disk;
struct blk_zone_wplug *zwplug;
unsigned int wp_offset;
- unsigned long flags;
/*
* Remember the capacity of the first sequential zone and check
@@ -2108,10 +2213,9 @@ static int blk_revalidate_seq_zone(struct blk_zone *zone, unsigned int idx,
if (!wp_offset || wp_offset >= zone->capacity)
return 0;
- zwplug = disk_get_and_lock_zone_wplug(disk, zone->wp, GFP_NOIO, &flags);
+ zwplug = disk_get_or_alloc_zone_wplug(disk, zone->wp, GFP_NOIO);
if (!zwplug)
return -ENOMEM;
- spin_unlock_irqrestore(&zwplug->lock, flags);
disk_put_zone_wplug(zwplug);
return 0;
@@ -2249,21 +2353,30 @@ int blk_revalidate_disk_zones(struct gendisk *disk)
}
memalloc_noio_restore(noio_flag);
+ if (ret <= 0)
+ goto free_resources;
+
/*
* If zones where reported, make sure that the entire disk capacity
* has been checked.
*/
- if (ret > 0 && args.sector != capacity) {
+ if (args.sector != capacity) {
pr_warn("%s: Missing zones from sector %llu\n",
disk->disk_name, args.sector);
ret = -ENODEV;
+ goto free_resources;
}
- if (ret > 0)
- return disk_update_zone_resources(disk, &args);
+ ret = disk_update_zone_resources(disk, &args);
+ if (ret)
+ goto free_resources;
+
+ return 0;
+free_resources:
pr_warn("%s: failed to revalidate zones\n", disk->disk_name);
+ kfree(args.zones_cond);
memflags = blk_mq_freeze_queue(q);
disk_free_zone_resources(disk);
blk_mq_unfreeze_queue(q, memflags);
diff --git a/block/blk.h b/block/blk.h
index c5b2115b9ea4..ec4674cdf2ea 100644
--- a/block/blk.h
+++ b/block/blk.h
@@ -55,7 +55,7 @@ bool __blk_freeze_queue_start(struct request_queue *q,
struct task_struct *owner);
int __bio_queue_enter(struct request_queue *q, struct bio *bio);
void submit_bio_noacct_nocheck(struct bio *bio, bool split);
-void bio_await_chain(struct bio *bio);
+int bio_submit_or_kill(struct bio *bio, unsigned int flags);
static inline bool blk_try_enter_queue(struct request_queue *q, bool pm)
{
@@ -108,11 +108,6 @@ static inline void blk_wait_io(struct completion *done)
struct block_device *blkdev_get_no_open(dev_t dev, bool autoload);
void blkdev_put_no_open(struct block_device *bdev);
-#define BIO_INLINE_VECS 4
-struct bio_vec *bvec_alloc(mempool_t *pool, unsigned short *nr_vecs,
- gfp_t gfp_mask);
-void bvec_free(mempool_t *pool, struct bio_vec *bv, unsigned short nr_vecs);
-
bool bvec_try_merge_hw_page(struct request_queue *q, struct bio_vec *bv,
struct page *page, unsigned len, unsigned offset);
diff --git a/block/bsg-lib.c b/block/bsg-lib.c
index 20cd0ef3c394..fdb4b290ca68 100644
--- a/block/bsg-lib.c
+++ b/block/bsg-lib.c
@@ -393,7 +393,7 @@ struct request_queue *bsg_setup_queue(struct device *dev, const char *name,
blk_queue_rq_timeout(q, BLK_DEFAULT_SG_TIMEOUT);
- bset->bd = bsg_register_queue(q, dev, name, bsg_transport_sg_io_fn);
+ bset->bd = bsg_register_queue(q, dev, name, bsg_transport_sg_io_fn, NULL);
if (IS_ERR(bset->bd)) {
ret = PTR_ERR(bset->bd);
goto out_cleanup_queue;
diff --git a/block/bsg.c b/block/bsg.c
index e0af6206ed28..82aaf3cee582 100644
--- a/block/bsg.c
+++ b/block/bsg.c
@@ -12,6 +12,7 @@
#include <linux/idr.h>
#include <linux/bsg.h>
#include <linux/slab.h>
+#include <linux/io_uring/cmd.h>
#include <scsi/scsi.h>
#include <scsi/scsi_ioctl.h>
@@ -28,6 +29,7 @@ struct bsg_device {
unsigned int timeout;
unsigned int reserved_size;
bsg_sg_io_fn *sg_io_fn;
+ bsg_uring_cmd_fn *uring_cmd_fn;
};
static inline struct bsg_device *to_bsg_device(struct inode *inode)
@@ -158,11 +160,38 @@ static long bsg_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
}
}
+static int bsg_check_uring_features(unsigned int issue_flags)
+{
+ /* BSG passthrough requires big SQE/CQE support */
+ if ((issue_flags & (IO_URING_F_SQE128|IO_URING_F_CQE32)) !=
+ (IO_URING_F_SQE128|IO_URING_F_CQE32))
+ return -EOPNOTSUPP;
+ return 0;
+}
+
+static int bsg_uring_cmd(struct io_uring_cmd *ioucmd, unsigned int issue_flags)
+{
+ struct bsg_device *bd = to_bsg_device(file_inode(ioucmd->file));
+ bool open_for_write = ioucmd->file->f_mode & FMODE_WRITE;
+ struct request_queue *q = bd->queue;
+ int ret;
+
+ ret = bsg_check_uring_features(issue_flags);
+ if (ret)
+ return ret;
+
+ if (!bd->uring_cmd_fn)
+ return -EOPNOTSUPP;
+
+ return bd->uring_cmd_fn(q, ioucmd, issue_flags, open_for_write);
+}
+
static const struct file_operations bsg_fops = {
.open = bsg_open,
.release = bsg_release,
.unlocked_ioctl = bsg_ioctl,
.compat_ioctl = compat_ptr_ioctl,
+ .uring_cmd = bsg_uring_cmd,
.owner = THIS_MODULE,
.llseek = default_llseek,
};
@@ -187,7 +216,8 @@ void bsg_unregister_queue(struct bsg_device *bd)
EXPORT_SYMBOL_GPL(bsg_unregister_queue);
struct bsg_device *bsg_register_queue(struct request_queue *q,
- struct device *parent, const char *name, bsg_sg_io_fn *sg_io_fn)
+ struct device *parent, const char *name, bsg_sg_io_fn *sg_io_fn,
+ bsg_uring_cmd_fn *uring_cmd_fn)
{
struct bsg_device *bd;
int ret;
@@ -199,6 +229,7 @@ struct bsg_device *bsg_register_queue(struct request_queue *q,
bd->reserved_size = INT_MAX;
bd->queue = q;
bd->sg_io_fn = sg_io_fn;
+ bd->uring_cmd_fn = uring_cmd_fn;
ret = ida_alloc_max(&bsg_minor_ida, BSG_MAX_DEVS - 1, GFP_KERNEL);
if (ret < 0) {
diff --git a/block/disk-events.c b/block/disk-events.c
index 9f9f9f8a2d6b..074731ecc3d2 100644
--- a/block/disk-events.c
+++ b/block/disk-events.c
@@ -290,13 +290,14 @@ EXPORT_SYMBOL(disk_check_media_change);
* Should be called when the media changes for @disk. Generates a uevent
* and attempts to free all dentries and inodes and invalidates all block
* device page cache entries in that case.
+ *
+ * Callers that need a partition re-scan should arrange for one explicitly.
*/
void disk_force_media_change(struct gendisk *disk)
{
disk_event_uevent(disk, DISK_EVENT_MEDIA_CHANGE);
inc_diskseq(disk);
bdev_mark_dead(disk->part0, true);
- set_bit(GD_NEED_PART_SCAN, &disk->state);
}
EXPORT_SYMBOL_GPL(disk_force_media_change);
diff --git a/block/ioctl.c b/block/ioctl.c
index 0b04661ac809..fc3be0549aa7 100644
--- a/block/ioctl.c
+++ b/block/ioctl.c
@@ -153,13 +153,7 @@ static int blk_ioctl_discard(struct block_device *bdev, blk_mode_t mode,
nr_sects = len >> SECTOR_SHIFT;
blk_start_plug(&plug);
- while (1) {
- if (fatal_signal_pending(current)) {
- if (prev)
- bio_await_chain(prev);
- err = -EINTR;
- goto out_unplug;
- }
+ while (!fatal_signal_pending(current)) {
bio = blk_alloc_discard_bio(bdev, &sector, &nr_sects,
GFP_KERNEL);
if (!bio)
@@ -167,12 +161,11 @@ static int blk_ioctl_discard(struct block_device *bdev, blk_mode_t mode,
prev = bio_chain_and_submit(prev, bio);
}
if (prev) {
- err = submit_bio_wait(prev);
+ err = bio_submit_or_kill(prev, BLKDEV_ZERO_KILLABLE);
if (err == -EOPNOTSUPP)
err = 0;
bio_put(prev);
}
-out_unplug:
blk_finish_plug(&plug);
fail:
filemap_invalidate_unlock(bdev->bd_mapping);
diff --git a/block/opal_proto.h b/block/opal_proto.h
index d247a457bf6e..7c24247aa186 100644
--- a/block/opal_proto.h
+++ b/block/opal_proto.h
@@ -19,6 +19,7 @@
enum {
TCG_SECP_00 = 0,
TCG_SECP_01,
+ TCG_SECP_02,
};
/*
@@ -125,6 +126,7 @@ enum opal_uid {
OPAL_LOCKING_INFO_TABLE,
OPAL_ENTERPRISE_LOCKING_INFO_TABLE,
OPAL_DATASTORE,
+ OPAL_LOCKING_TABLE,
/* C_PIN_TABLE object ID's */
OPAL_C_PIN_MSID,
OPAL_C_PIN_SID,
@@ -154,6 +156,7 @@ enum opal_method {
OPAL_AUTHENTICATE,
OPAL_RANDOM,
OPAL_ERASE,
+ OPAL_REACTIVATE,
};
enum opal_token {
@@ -224,6 +227,8 @@ enum opal_lockingstate {
enum opal_parameter {
OPAL_SUM_SET_LIST = 0x060000,
+ OPAL_SUM_RANGE_POLICY = 0x060001,
+ OPAL_SUM_ADMIN1_PIN = 0x060002,
};
enum opal_revertlsp {
@@ -269,6 +274,25 @@ struct opal_header {
struct opal_data_subpacket subpkt;
};
+/*
+ * TCG_Storage_Architecture_Core_Spec_v2.01_r1.00
+ * Section: 3.3.4.7.5 STACK_RESET
+ */
+#define OPAL_STACK_RESET 0x0002
+
+struct opal_stack_reset {
+ u8 extendedComID[4];
+ __be32 request_code;
+};
+
+struct opal_stack_reset_response {
+ u8 extendedComID[4];
+ __be32 request_code;
+ u8 reserved0[2];
+ __be16 data_length;
+ __be32 response;
+};
+
#define FC_TPER 0x0001
#define FC_LOCKING 0x0002
#define FC_GEOMETRY 0x0003
diff --git a/block/partitions/acorn.c b/block/partitions/acorn.c
index d2fc122d7426..9f7389f174d0 100644
--- a/block/partitions/acorn.c
+++ b/block/partitions/acorn.c
@@ -40,9 +40,7 @@ adfs_partition(struct parsed_partitions *state, char *name, char *data,
(le32_to_cpu(dr->disc_size) >> 9);
if (name) {
- strlcat(state->pp_buf, " [", PAGE_SIZE);
- strlcat(state->pp_buf, name, PAGE_SIZE);
- strlcat(state->pp_buf, "]", PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, " [%s]", name);
}
put_partition(state, slot, first_sector, nr_sects);
return dr;
@@ -78,14 +76,14 @@ static int riscix_partition(struct parsed_partitions *state,
if (!rr)
return -1;
- strlcat(state->pp_buf, " [RISCiX]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [RISCiX]");
if (rr->magic == RISCIX_MAGIC) {
unsigned long size = nr_sects > 2 ? 2 : nr_sects;
int part;
- strlcat(state->pp_buf, " <", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " <");
put_partition(state, slot++, first_sect, size);
for (part = 0; part < 8; part++) {
@@ -94,13 +92,11 @@ static int riscix_partition(struct parsed_partitions *state,
put_partition(state, slot++,
le32_to_cpu(rr->part[part].start),
le32_to_cpu(rr->part[part].length));
- strlcat(state->pp_buf, "(", PAGE_SIZE);
- strlcat(state->pp_buf, rr->part[part].name, PAGE_SIZE);
- strlcat(state->pp_buf, ")", PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "(%s)", rr->part[part].name);
}
}
- strlcat(state->pp_buf, " >\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >\n");
} else {
put_partition(state, slot++, first_sect, nr_sects);
}
@@ -130,7 +126,7 @@ static int linux_partition(struct parsed_partitions *state,
struct linux_part *linuxp;
unsigned long size = nr_sects > 2 ? 2 : nr_sects;
- strlcat(state->pp_buf, " [Linux]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [Linux]");
put_partition(state, slot++, first_sect, size);
@@ -138,7 +134,7 @@ static int linux_partition(struct parsed_partitions *state,
if (!linuxp)
return -1;
- strlcat(state->pp_buf, " <", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " <");
while (linuxp->magic == cpu_to_le32(LINUX_NATIVE_MAGIC) ||
linuxp->magic == cpu_to_le32(LINUX_SWAP_MAGIC)) {
if (slot == state->limit)
@@ -148,7 +144,7 @@ static int linux_partition(struct parsed_partitions *state,
le32_to_cpu(linuxp->nr_sects));
linuxp ++;
}
- strlcat(state->pp_buf, " >", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >");
put_dev_sector(sect);
return slot;
@@ -293,7 +289,7 @@ int adfspart_check_ADFS(struct parsed_partitions *state)
break;
}
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
#endif
@@ -366,7 +362,7 @@ int adfspart_check_ICS(struct parsed_partitions *state)
return 0;
}
- strlcat(state->pp_buf, " [ICS]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [ICS]");
for (slot = 1, p = (const struct ics_part *)data; p->size; p++) {
u32 start = le32_to_cpu(p->start);
@@ -400,7 +396,7 @@ int adfspart_check_ICS(struct parsed_partitions *state)
}
put_dev_sector(sect);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
#endif
@@ -460,7 +456,7 @@ int adfspart_check_POWERTEC(struct parsed_partitions *state)
return 0;
}
- strlcat(state->pp_buf, " [POWERTEC]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [POWERTEC]");
for (i = 0, p = (const struct ptec_part *)data; i < 12; i++, p++) {
u32 start = le32_to_cpu(p->start);
@@ -471,7 +467,7 @@ int adfspart_check_POWERTEC(struct parsed_partitions *state)
}
put_dev_sector(sect);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
#endif
@@ -542,7 +538,7 @@ int adfspart_check_EESOX(struct parsed_partitions *state)
size = get_capacity(state->disk);
put_partition(state, slot++, start, size - start);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
}
return i ? 1 : 0;
diff --git a/block/partitions/aix.c b/block/partitions/aix.c
index a886cefbefbb..29b8f4cebb63 100644
--- a/block/partitions/aix.c
+++ b/block/partitions/aix.c
@@ -173,24 +173,22 @@ int aix_partition(struct parsed_partitions *state)
if (d) {
struct lvm_rec *p = (struct lvm_rec *)d;
u16 lvm_version = be16_to_cpu(p->version);
- char tmp[64];
if (lvm_version == 1) {
int pp_size_log2 = be16_to_cpu(p->pp_size);
pp_bytes_size = 1 << pp_size_log2;
pp_blocks_size = pp_bytes_size / 512;
- snprintf(tmp, sizeof(tmp),
- " AIX LVM header version %u found\n",
- lvm_version);
+ seq_buf_printf(&state->pp_buf,
+ " AIX LVM header version %u found\n",
+ lvm_version);
vgda_len = be32_to_cpu(p->vgda_len);
vgda_sector = be32_to_cpu(p->vgda_psn[0]);
} else {
- snprintf(tmp, sizeof(tmp),
- " unsupported AIX LVM version %d found\n",
- lvm_version);
+ seq_buf_printf(&state->pp_buf,
+ " unsupported AIX LVM version %d found\n",
+ lvm_version);
}
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
put_dev_sector(sect);
}
if (vgda_sector && (d = read_part_sector(state, vgda_sector, &sect))) {
@@ -251,14 +249,11 @@ int aix_partition(struct parsed_partitions *state)
continue;
}
if (lp_ix == lvip[lv_ix].pps_per_lv) {
- char tmp[70];
-
put_partition(state, lv_ix + 1,
(i + 1 - lp_ix) * pp_blocks_size + psn_part1,
lvip[lv_ix].pps_per_lv * pp_blocks_size);
- snprintf(tmp, sizeof(tmp), " <%s>\n",
- n[lv_ix].name);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, " <%s>\n",
+ n[lv_ix].name);
lvip[lv_ix].lv_is_contiguous = 1;
ret = 1;
next_lp_ix = 1;
diff --git a/block/partitions/amiga.c b/block/partitions/amiga.c
index 506921095412..8325046a14eb 100644
--- a/block/partitions/amiga.c
+++ b/block/partitions/amiga.c
@@ -81,13 +81,8 @@ int amiga_partition(struct parsed_partitions *state)
/* blksize is blocks per 512 byte standard block */
blksize = be32_to_cpu( rdb->rdb_BlockBytes ) / 512;
- {
- char tmp[7 + 10 + 1 + 1];
-
- /* Be more informative */
- snprintf(tmp, sizeof(tmp), " RDSK (%d)", blksize * 512);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
- }
+ /* Be more informative */
+ seq_buf_printf(&state->pp_buf, " RDSK (%d)", blksize * 512);
blk = be32_to_cpu(rdb->rdb_PartitionList);
put_dev_sector(sect);
for (part = 1; (s32) blk>0 && part<=16; part++, put_dev_sector(sect)) {
@@ -179,27 +174,27 @@ int amiga_partition(struct parsed_partitions *state)
{
/* Be even more informative to aid mounting */
char dostype[4];
- char tmp[42];
__be32 *dt = (__be32 *)dostype;
*dt = pb->pb_Environment[16];
if (dostype[3] < ' ')
- snprintf(tmp, sizeof(tmp), " (%c%c%c^%c)",
- dostype[0], dostype[1],
- dostype[2], dostype[3] + '@' );
+ seq_buf_printf(&state->pp_buf,
+ " (%c%c%c^%c)",
+ dostype[0], dostype[1],
+ dostype[2],
+ dostype[3] + '@');
else
- snprintf(tmp, sizeof(tmp), " (%c%c%c%c)",
- dostype[0], dostype[1],
- dostype[2], dostype[3]);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
- snprintf(tmp, sizeof(tmp), "(res %d spb %d)",
- be32_to_cpu(pb->pb_Environment[6]),
- be32_to_cpu(pb->pb_Environment[4]));
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf,
+ " (%c%c%c%c)",
+ dostype[0], dostype[1],
+ dostype[2], dostype[3]);
+ seq_buf_printf(&state->pp_buf, "(res %d spb %d)",
+ be32_to_cpu(pb->pb_Environment[6]),
+ be32_to_cpu(pb->pb_Environment[4]));
}
res = 1;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
rdb_done:
return res;
diff --git a/block/partitions/atari.c b/block/partitions/atari.c
index 9655c728262a..2438d1448f38 100644
--- a/block/partitions/atari.c
+++ b/block/partitions/atari.c
@@ -70,7 +70,7 @@ int atari_partition(struct parsed_partitions *state)
}
pi = &rs->part[0];
- strlcat(state->pp_buf, " AHDI", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " AHDI");
for (slot = 1; pi < &rs->part[4] && slot < state->limit; slot++, pi++) {
struct rootsector *xrs;
Sector sect2;
@@ -89,7 +89,7 @@ int atari_partition(struct parsed_partitions *state)
#ifdef ICD_PARTS
part_fmt = 1;
#endif
- strlcat(state->pp_buf, " XGM<", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " XGM<");
partsect = extensect = be32_to_cpu(pi->st);
while (1) {
xrs = read_part_sector(state, partsect, &sect2);
@@ -128,14 +128,14 @@ int atari_partition(struct parsed_partitions *state)
break;
}
}
- strlcat(state->pp_buf, " >", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >");
}
#ifdef ICD_PARTS
if ( part_fmt!=1 ) { /* no extended partitions -> test ICD-format */
pi = &rs->icdpart[0];
/* sanity check: no ICD format if first partition invalid */
if (OK_id(pi->id)) {
- strlcat(state->pp_buf, " ICD<", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " ICD<");
for (; pi < &rs->icdpart[8] && slot < state->limit; slot++, pi++) {
/* accept only GEM,BGM,RAW,LNX,SWP partitions */
if (!((pi->flg & 1) && OK_id(pi->id)))
@@ -144,13 +144,13 @@ int atari_partition(struct parsed_partitions *state)
be32_to_cpu(pi->st),
be32_to_cpu(pi->siz));
}
- strlcat(state->pp_buf, " >", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >");
}
}
#endif
put_dev_sector(sect);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
diff --git a/block/partitions/check.h b/block/partitions/check.h
index e5c1c61eb353..b0997467b61a 100644
--- a/block/partitions/check.h
+++ b/block/partitions/check.h
@@ -1,6 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
#include <linux/pagemap.h>
#include <linux/blkdev.h>
+#include <linux/seq_buf.h>
#include "../blk.h"
/*
@@ -20,7 +21,7 @@ struct parsed_partitions {
int next;
int limit;
bool access_beyond_eod;
- char *pp_buf;
+ struct seq_buf pp_buf;
};
typedef struct {
@@ -37,12 +38,9 @@ static inline void
put_partition(struct parsed_partitions *p, int n, sector_t from, sector_t size)
{
if (n < p->limit) {
- char tmp[1 + BDEVNAME_SIZE + 10 + 1];
-
p->parts[n].from = from;
p->parts[n].size = size;
- snprintf(tmp, sizeof(tmp), " %s%d", p->name, n);
- strlcat(p->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&p->pp_buf, " %s%d", p->name, n);
}
}
diff --git a/block/partitions/cmdline.c b/block/partitions/cmdline.c
index a2b1870c3fd4..4fd52ed154b4 100644
--- a/block/partitions/cmdline.c
+++ b/block/partitions/cmdline.c
@@ -229,7 +229,6 @@ static int add_part(int slot, struct cmdline_subpart *subpart,
struct parsed_partitions *state)
{
struct partition_meta_info *info;
- char tmp[sizeof(info->volname) + 4];
if (slot >= state->limit)
return 1;
@@ -244,8 +243,7 @@ static int add_part(int slot, struct cmdline_subpart *subpart,
strscpy(info->volname, subpart->name, sizeof(info->volname));
- snprintf(tmp, sizeof(tmp), "(%s)", info->volname);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "(%s)", info->volname);
state->parts[slot].has_info = true;
@@ -379,7 +377,7 @@ int cmdline_partition(struct parsed_partitions *state)
cmdline_parts_set(parts, disk_size, state);
cmdline_parts_verifier(1, state);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
diff --git a/block/partitions/core.c b/block/partitions/core.c
index 740228750aaf..5d5332ce586b 100644
--- a/block/partitions/core.c
+++ b/block/partitions/core.c
@@ -8,6 +8,7 @@
#include <linux/major.h>
#include <linux/slab.h>
#include <linux/string.h>
+#include <linux/sysfs.h>
#include <linux/ctype.h>
#include <linux/vmalloc.h>
#include <linux/raid/detect.h>
@@ -123,16 +124,16 @@ static struct parsed_partitions *check_partition(struct gendisk *hd)
state = allocate_partitions(hd);
if (!state)
return NULL;
- state->pp_buf = (char *)__get_free_page(GFP_KERNEL);
- if (!state->pp_buf) {
+ state->pp_buf.buffer = (char *)__get_free_page(GFP_KERNEL);
+ if (!state->pp_buf.buffer) {
free_partitions(state);
return NULL;
}
- state->pp_buf[0] = '\0';
+ seq_buf_init(&state->pp_buf, state->pp_buf.buffer, PAGE_SIZE);
state->disk = hd;
strscpy(state->name, hd->disk_name);
- snprintf(state->pp_buf, PAGE_SIZE, " %s:", state->name);
+ seq_buf_printf(&state->pp_buf, " %s:", state->name);
if (isdigit(state->name[strlen(state->name)-1]))
sprintf(state->name, "p");
@@ -151,9 +152,9 @@ static struct parsed_partitions *check_partition(struct gendisk *hd)
}
if (res > 0) {
- printk(KERN_INFO "%s", state->pp_buf);
+ printk(KERN_INFO "%s", seq_buf_str(&state->pp_buf));
- free_page((unsigned long)state->pp_buf);
+ free_page((unsigned long)state->pp_buf.buffer);
return state;
}
if (state->access_beyond_eod)
@@ -164,12 +165,12 @@ static struct parsed_partitions *check_partition(struct gendisk *hd)
if (err)
res = err;
if (res) {
- strlcat(state->pp_buf,
- " unable to read partition table\n", PAGE_SIZE);
- printk(KERN_INFO "%s", state->pp_buf);
+ seq_buf_puts(&state->pp_buf,
+ " unable to read partition table\n");
+ printk(KERN_INFO "%s", seq_buf_str(&state->pp_buf));
}
- free_page((unsigned long)state->pp_buf);
+ free_page((unsigned long)state->pp_buf.buffer);
free_partitions(state);
return ERR_PTR(res);
}
@@ -177,31 +178,31 @@ static struct parsed_partitions *check_partition(struct gendisk *hd)
static ssize_t part_partition_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
- return sprintf(buf, "%d\n", bdev_partno(dev_to_bdev(dev)));
+ return sysfs_emit(buf, "%d\n", bdev_partno(dev_to_bdev(dev)));
}
static ssize_t part_start_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
- return sprintf(buf, "%llu\n", dev_to_bdev(dev)->bd_start_sect);
+ return sysfs_emit(buf, "%llu\n", dev_to_bdev(dev)->bd_start_sect);
}
static ssize_t part_ro_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
- return sprintf(buf, "%d\n", bdev_read_only(dev_to_bdev(dev)));
+ return sysfs_emit(buf, "%d\n", bdev_read_only(dev_to_bdev(dev)));
}
static ssize_t part_alignment_offset_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
- return sprintf(buf, "%u\n", bdev_alignment_offset(dev_to_bdev(dev)));
+ return sysfs_emit(buf, "%u\n", bdev_alignment_offset(dev_to_bdev(dev)));
}
static ssize_t part_discard_alignment_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
- return sprintf(buf, "%u\n", bdev_discard_alignment(dev_to_bdev(dev)));
+ return sysfs_emit(buf, "%u\n", bdev_discard_alignment(dev_to_bdev(dev)));
}
static DEVICE_ATTR(partition, 0444, part_partition_show, NULL);
diff --git a/block/partitions/efi.c b/block/partitions/efi.c
index 75474fb3848e..9865d59093fa 100644
--- a/block/partitions/efi.c
+++ b/block/partitions/efi.c
@@ -751,6 +751,6 @@ int efi_partition(struct parsed_partitions *state)
}
kfree(ptes);
kfree(gpt);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
diff --git a/block/partitions/ibm.c b/block/partitions/ibm.c
index 9311ad5fb95d..54047e722a9d 100644
--- a/block/partitions/ibm.c
+++ b/block/partitions/ibm.c
@@ -173,15 +173,13 @@ static int find_vol1_partitions(struct parsed_partitions *state,
{
sector_t blk;
int counter;
- char tmp[64];
Sector sect;
unsigned char *data;
loff_t offset, size;
struct vtoc_format1_label f1;
int secperblk;
- snprintf(tmp, sizeof(tmp), "VOL1/%8s:", name);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "VOL1/%8s:", name);
/*
* get start of VTOC from the disk label and then search for format1
* and format8 labels
@@ -219,7 +217,7 @@ static int find_vol1_partitions(struct parsed_partitions *state,
blk++;
data = read_part_sector(state, blk * secperblk, &sect);
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
if (!data)
return -1;
@@ -237,11 +235,9 @@ static int find_lnx1_partitions(struct parsed_partitions *state,
dasd_information2_t *info)
{
loff_t offset, geo_size, size;
- char tmp[64];
int secperblk;
- snprintf(tmp, sizeof(tmp), "LNX1/%8s:", name);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "LNX1/%8s:", name);
secperblk = blocksize >> 9;
if (label->lnx.ldl_version == 0xf2) {
size = label->lnx.formatted_blocks * secperblk;
@@ -258,7 +254,7 @@ static int find_lnx1_partitions(struct parsed_partitions *state,
size = nr_sectors;
if (size != geo_size) {
if (!info) {
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
if (!strcmp(info->type, "ECKD"))
@@ -270,7 +266,7 @@ static int find_lnx1_partitions(struct parsed_partitions *state,
/* first and only partition starts in the first block after the label */
offset = labelsect + secperblk;
put_partition(state, 1, offset, size - offset);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
@@ -282,7 +278,6 @@ static int find_cms1_partitions(struct parsed_partitions *state,
sector_t labelsect)
{
loff_t offset, size;
- char tmp[64];
int secperblk;
/*
@@ -291,14 +286,12 @@ static int find_cms1_partitions(struct parsed_partitions *state,
blocksize = label->cms.block_size;
secperblk = blocksize >> 9;
if (label->cms.disk_offset != 0) {
- snprintf(tmp, sizeof(tmp), "CMS1/%8s(MDSK):", name);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "CMS1/%8s(MDSK):", name);
/* disk is reserved minidisk */
offset = label->cms.disk_offset * secperblk;
size = (label->cms.block_count - 1) * secperblk;
} else {
- snprintf(tmp, sizeof(tmp), "CMS1/%8s:", name);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "CMS1/%8s:", name);
/*
* Special case for FBA devices:
* If an FBA device is CMS formatted with blocksize > 512 byte
@@ -314,7 +307,7 @@ static int find_cms1_partitions(struct parsed_partitions *state,
}
put_partition(state, 1, offset, size-offset);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
@@ -391,11 +384,11 @@ int ibm_partition(struct parsed_partitions *state)
*/
res = 1;
if (info->format == DASD_FORMAT_LDL) {
- strlcat(state->pp_buf, "(nonl)", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "(nonl)");
size = nr_sectors;
offset = (info->label_block + 1) * (blocksize >> 9);
put_partition(state, 1, offset, size-offset);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
}
} else
res = 0;
diff --git a/block/partitions/karma.c b/block/partitions/karma.c
index 4d93512f4bd4..a4e3c5050177 100644
--- a/block/partitions/karma.c
+++ b/block/partitions/karma.c
@@ -53,7 +53,7 @@ int karma_partition(struct parsed_partitions *state)
}
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
}
diff --git a/block/partitions/ldm.c b/block/partitions/ldm.c
index 776b4ad95091..c0bdcae58a3e 100644
--- a/block/partitions/ldm.c
+++ b/block/partitions/ldm.c
@@ -582,7 +582,7 @@ static bool ldm_create_data_partitions (struct parsed_partitions *pp,
return false;
}
- strlcat(pp->pp_buf, " [LDM]", PAGE_SIZE);
+ seq_buf_puts(&pp->pp_buf, " [LDM]");
/* Create the data partitions */
list_for_each (item, &ldb->v_part) {
@@ -597,7 +597,7 @@ static bool ldm_create_data_partitions (struct parsed_partitions *pp,
part_num++;
}
- strlcat(pp->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&pp->pp_buf, "\n");
return true;
}
diff --git a/block/partitions/mac.c b/block/partitions/mac.c
index b02530d98629..df03ca428e15 100644
--- a/block/partitions/mac.c
+++ b/block/partitions/mac.c
@@ -86,7 +86,7 @@ int mac_partition(struct parsed_partitions *state)
if (blocks_in_map >= state->limit)
blocks_in_map = state->limit - 1;
- strlcat(state->pp_buf, " [mac]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [mac]");
for (slot = 1; slot <= blocks_in_map; ++slot) {
int pos = slot * secsize;
put_dev_sector(sect);
@@ -152,6 +152,6 @@ int mac_partition(struct parsed_partitions *state)
#endif
put_dev_sector(sect);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
diff --git a/block/partitions/msdos.c b/block/partitions/msdos.c
index 073be78ba0b0..200ea53ea6a2 100644
--- a/block/partitions/msdos.c
+++ b/block/partitions/msdos.c
@@ -263,18 +263,11 @@ static void parse_solaris_x86(struct parsed_partitions *state,
put_dev_sector(sect);
return;
}
- {
- char tmp[1 + BDEVNAME_SIZE + 10 + 11 + 1];
-
- snprintf(tmp, sizeof(tmp), " %s%d: <solaris:", state->name, origin);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
- }
+ seq_buf_printf(&state->pp_buf, " %s%d: <solaris:", state->name, origin);
if (le32_to_cpu(v->v_version) != 1) {
- char tmp[64];
-
- snprintf(tmp, sizeof(tmp), " cannot handle version %d vtoc>\n",
- le32_to_cpu(v->v_version));
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf,
+ " cannot handle version %d vtoc>\n",
+ le32_to_cpu(v->v_version));
put_dev_sector(sect);
return;
}
@@ -282,12 +275,10 @@ static void parse_solaris_x86(struct parsed_partitions *state,
max_nparts = le16_to_cpu(v->v_nparts) > 8 ? SOLARIS_X86_NUMSLICE : 8;
for (i = 0; i < max_nparts && state->next < state->limit; i++) {
struct solaris_x86_slice *s = &v->v_slice[i];
- char tmp[3 + 10 + 1 + 1];
if (s->s_size == 0)
continue;
- snprintf(tmp, sizeof(tmp), " [s%d]", i);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, " [s%d]", i);
/* solaris partitions are relative to current MS-DOS
* one; must add the offset of the current partition */
put_partition(state, state->next++,
@@ -295,7 +286,7 @@ static void parse_solaris_x86(struct parsed_partitions *state,
le32_to_cpu(s->s_size));
}
put_dev_sector(sect);
- strlcat(state->pp_buf, " >\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >\n");
#endif
}
@@ -359,7 +350,6 @@ static void parse_bsd(struct parsed_partitions *state,
Sector sect;
struct bsd_disklabel *l;
struct bsd_partition *p;
- char tmp[64];
l = read_part_sector(state, offset + 1, &sect);
if (!l)
@@ -369,8 +359,7 @@ static void parse_bsd(struct parsed_partitions *state,
return;
}
- snprintf(tmp, sizeof(tmp), " %s%d: <%s:", state->name, origin, flavour);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, " %s%d: <%s:", state->name, origin, flavour);
if (le16_to_cpu(l->d_npartitions) < max_partitions)
max_partitions = le16_to_cpu(l->d_npartitions);
@@ -391,18 +380,16 @@ static void parse_bsd(struct parsed_partitions *state,
/* full parent partition, we have it already */
continue;
if (offset > bsd_start || offset+size < bsd_start+bsd_size) {
- strlcat(state->pp_buf, "bad subpartition - ignored\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "bad subpartition - ignored\n");
continue;
}
put_partition(state, state->next++, bsd_start, bsd_size);
}
put_dev_sector(sect);
- if (le16_to_cpu(l->d_npartitions) > max_partitions) {
- snprintf(tmp, sizeof(tmp), " (ignored %d more)",
- le16_to_cpu(l->d_npartitions) - max_partitions);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
- }
- strlcat(state->pp_buf, " >\n", PAGE_SIZE);
+ if (le16_to_cpu(l->d_npartitions) > max_partitions)
+ seq_buf_printf(&state->pp_buf, " (ignored %d more)",
+ le16_to_cpu(l->d_npartitions) - max_partitions);
+ seq_buf_puts(&state->pp_buf, " >\n");
}
#endif
@@ -496,12 +483,7 @@ static void parse_unixware(struct parsed_partitions *state,
put_dev_sector(sect);
return;
}
- {
- char tmp[1 + BDEVNAME_SIZE + 10 + 12 + 1];
-
- snprintf(tmp, sizeof(tmp), " %s%d: <unixware:", state->name, origin);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
- }
+ seq_buf_printf(&state->pp_buf, " %s%d: <unixware:", state->name, origin);
p = &l->vtoc.v_slice[1];
/* I omit the 0th slice as it is the same as whole disk. */
while (p - &l->vtoc.v_slice[0] < UNIXWARE_NUMSLICE) {
@@ -515,7 +497,7 @@ static void parse_unixware(struct parsed_partitions *state,
p++;
}
put_dev_sector(sect);
- strlcat(state->pp_buf, " >\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >\n");
#endif
}
@@ -546,10 +528,7 @@ static void parse_minix(struct parsed_partitions *state,
* the normal boot sector. */
if (msdos_magic_present(data + 510) &&
p->sys_ind == MINIX_PARTITION) { /* subpartition table present */
- char tmp[1 + BDEVNAME_SIZE + 10 + 9 + 1];
-
- snprintf(tmp, sizeof(tmp), " %s%d: <minix:", state->name, origin);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, " %s%d: <minix:", state->name, origin);
for (i = 0; i < MINIX_NR_SUBPARTITIONS; i++, p++) {
if (state->next == state->limit)
break;
@@ -558,7 +537,7 @@ static void parse_minix(struct parsed_partitions *state,
put_partition(state, state->next++,
start_sect(p), nr_sects(p));
}
- strlcat(state->pp_buf, " >\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >\n");
}
put_dev_sector(sect);
#endif /* CONFIG_MINIX_SUBPARTITION */
@@ -602,7 +581,7 @@ int msdos_partition(struct parsed_partitions *state)
#ifdef CONFIG_AIX_PARTITION
return aix_partition(state);
#else
- strlcat(state->pp_buf, " [AIX]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " [AIX]");
return 0;
#endif
}
@@ -629,7 +608,7 @@ int msdos_partition(struct parsed_partitions *state)
fb = (struct fat_boot_sector *) data;
if (slot == 1 && fb->reserved && fb->fats
&& fat_valid_media(fb->media)) {
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
} else {
@@ -678,9 +657,9 @@ int msdos_partition(struct parsed_partitions *state)
n = min(size, max(sector_size, n));
put_partition(state, slot, start, n);
- strlcat(state->pp_buf, " <", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " <");
parse_extended(state, start, size, disksig);
- strlcat(state->pp_buf, " >", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, " >");
continue;
}
put_partition(state, slot, start, size);
@@ -688,12 +667,12 @@ int msdos_partition(struct parsed_partitions *state)
if (p->sys_ind == LINUX_RAID_PARTITION)
state->parts[slot].flags = ADDPART_FLAG_RAID;
if (p->sys_ind == DM6_PARTITION)
- strlcat(state->pp_buf, "[DM]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "[DM]");
if (p->sys_ind == EZD_PARTITION)
- strlcat(state->pp_buf, "[EZD]", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "[EZD]");
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
/* second pass - output for each on a separate line */
p = (struct msdos_partition *) (0x1be + data);
diff --git a/block/partitions/of.c b/block/partitions/of.c
index 4e760fdffb3f..c22b60661098 100644
--- a/block/partitions/of.c
+++ b/block/partitions/of.c
@@ -36,7 +36,6 @@ static void add_of_partition(struct parsed_partitions *state, int slot,
struct device_node *np)
{
struct partition_meta_info *info;
- char tmp[sizeof(info->volname) + 4];
const char *partname;
int len;
@@ -63,8 +62,7 @@ static void add_of_partition(struct parsed_partitions *state, int slot,
partname = of_get_property(np, "name", &len);
strscpy(info->volname, partname, sizeof(info->volname));
- snprintf(tmp, sizeof(tmp), "(%s)", info->volname);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "(%s)", info->volname);
}
int of_partition(struct parsed_partitions *state)
@@ -104,7 +102,7 @@ int of_partition(struct parsed_partitions *state)
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
}
diff --git a/block/partitions/osf.c b/block/partitions/osf.c
index 84560d0765ed..2a692584dba9 100644
--- a/block/partitions/osf.c
+++ b/block/partitions/osf.c
@@ -81,7 +81,7 @@ int osf_partition(struct parsed_partitions *state)
le32_to_cpu(partition->p_size));
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
}
diff --git a/block/partitions/sgi.c b/block/partitions/sgi.c
index b5ecddd5181a..2383ca63cd66 100644
--- a/block/partitions/sgi.c
+++ b/block/partitions/sgi.c
@@ -79,7 +79,7 @@ int sgi_partition(struct parsed_partitions *state)
}
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
}
diff --git a/block/partitions/sun.c b/block/partitions/sun.c
index 2419af76120f..92c645fcd2e0 100644
--- a/block/partitions/sun.c
+++ b/block/partitions/sun.c
@@ -121,7 +121,7 @@ int sun_partition(struct parsed_partitions *state)
}
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
}
diff --git a/block/partitions/sysv68.c b/block/partitions/sysv68.c
index 6f6257fd4eb4..470e0f9de7be 100644
--- a/block/partitions/sysv68.c
+++ b/block/partitions/sysv68.c
@@ -54,7 +54,6 @@ int sysv68_partition(struct parsed_partitions *state)
unsigned char *data;
struct dkblk0 *b;
struct slice *slice;
- char tmp[64];
data = read_part_sector(state, 0, &sect);
if (!data)
@@ -74,8 +73,7 @@ int sysv68_partition(struct parsed_partitions *state)
return -1;
slices -= 1; /* last slice is the whole disk */
- snprintf(tmp, sizeof(tmp), "sysV68: %s(s%u)", state->name, slices);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "sysV68: %s(s%u)", state->name, slices);
slice = (struct slice *)data;
for (i = 0; i < slices; i++, slice++) {
if (slot == state->limit)
@@ -84,12 +82,11 @@ int sysv68_partition(struct parsed_partitions *state)
put_partition(state, slot,
be32_to_cpu(slice->blkoff),
be32_to_cpu(slice->nblocks));
- snprintf(tmp, sizeof(tmp), "(s%u)", i);
- strlcat(state->pp_buf, tmp, PAGE_SIZE);
+ seq_buf_printf(&state->pp_buf, "(s%u)", i);
}
slot++;
}
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
put_dev_sector(sect);
return 1;
}
diff --git a/block/partitions/ultrix.c b/block/partitions/ultrix.c
index 4aaa81043ca0..b4b9ddc57a5d 100644
--- a/block/partitions/ultrix.c
+++ b/block/partitions/ultrix.c
@@ -39,7 +39,7 @@ int ultrix_partition(struct parsed_partitions *state)
label->pt_part[i].pi_blkoff,
label->pt_part[i].pi_nblocks);
put_dev_sector(sect);
- strlcat(state->pp_buf, "\n", PAGE_SIZE);
+ seq_buf_puts(&state->pp_buf, "\n");
return 1;
} else {
put_dev_sector(sect);
diff --git a/block/sed-opal.c b/block/sed-opal.c
index 3ded1ca723ca..79b290d9458a 100644
--- a/block/sed-opal.c
+++ b/block/sed-opal.c
@@ -160,6 +160,8 @@ static const u8 opaluid[][OPAL_UID_LENGTH] = {
{ 0x00, 0x00, 0x08, 0x01, 0x00, 0x00, 0x00, 0x00 },
[OPAL_DATASTORE] =
{ 0x00, 0x00, 0x10, 0x01, 0x00, 0x00, 0x00, 0x00 },
+ [OPAL_LOCKING_TABLE] =
+ { 0x00, 0x00, 0x08, 0x02, 0x00, 0x00, 0x00, 0x00 },
/* C_PIN_TABLE object ID's */
[OPAL_C_PIN_MSID] =
@@ -218,6 +220,8 @@ static const u8 opalmethod[][OPAL_METHOD_LENGTH] = {
{ 0x00, 0x00, 0x00, 0x06, 0x00, 0x00, 0x06, 0x01 },
[OPAL_ERASE] =
{ 0x00, 0x00, 0x00, 0x06, 0x00, 0x00, 0x08, 0x03 },
+ [OPAL_REACTIVATE] =
+ { 0x00, 0x00, 0x00, 0x06, 0x00, 0x00, 0x08, 0x01 },
};
static int end_opal_session_error(struct opal_dev *dev);
@@ -1514,7 +1518,7 @@ static inline int enable_global_lr(struct opal_dev *dev, u8 *uid,
return err;
}
-static int setup_locking_range(struct opal_dev *dev, void *data)
+static int setup_enable_range(struct opal_dev *dev, void *data)
{
u8 uid[OPAL_UID_LENGTH];
struct opal_user_lr_setup *setup = data;
@@ -1528,38 +1532,47 @@ static int setup_locking_range(struct opal_dev *dev, void *data)
if (lr == 0)
err = enable_global_lr(dev, uid, setup);
- else {
- err = cmd_start(dev, uid, opalmethod[OPAL_SET]);
+ else
+ err = generic_lr_enable_disable(dev, uid, !!setup->RLE, !!setup->WLE, 0, 0);
+ if (err) {
+ pr_debug("Failed to create enable lr command.\n");
+ return err;
+ }
- add_token_u8(&err, dev, OPAL_STARTNAME);
- add_token_u8(&err, dev, OPAL_VALUES);
- add_token_u8(&err, dev, OPAL_STARTLIST);
+ return finalize_and_send(dev, parse_and_check_status);
+}
- add_token_u8(&err, dev, OPAL_STARTNAME);
- add_token_u8(&err, dev, OPAL_RANGESTART);
- add_token_u64(&err, dev, setup->range_start);
- add_token_u8(&err, dev, OPAL_ENDNAME);
+static int setup_locking_range_start_length(struct opal_dev *dev, void *data)
+{
+ int err;
+ u8 uid[OPAL_UID_LENGTH];
+ struct opal_user_lr_setup *setup = data;
- add_token_u8(&err, dev, OPAL_STARTNAME);
- add_token_u8(&err, dev, OPAL_RANGELENGTH);
- add_token_u64(&err, dev, setup->range_length);
- add_token_u8(&err, dev, OPAL_ENDNAME);
+ err = build_locking_range(uid, sizeof(uid), setup->session.opal_key.lr);
+ if (err)
+ return err;
- add_token_u8(&err, dev, OPAL_STARTNAME);
- add_token_u8(&err, dev, OPAL_READLOCKENABLED);
- add_token_u64(&err, dev, !!setup->RLE);
- add_token_u8(&err, dev, OPAL_ENDNAME);
+ err = cmd_start(dev, uid, opalmethod[OPAL_SET]);
- add_token_u8(&err, dev, OPAL_STARTNAME);
- add_token_u8(&err, dev, OPAL_WRITELOCKENABLED);
- add_token_u64(&err, dev, !!setup->WLE);
- add_token_u8(&err, dev, OPAL_ENDNAME);
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u8(&err, dev, OPAL_VALUES);
+ add_token_u8(&err, dev, OPAL_STARTLIST);
+
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u8(&err, dev, OPAL_RANGESTART);
+ add_token_u64(&err, dev, setup->range_start);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u8(&err, dev, OPAL_RANGELENGTH);
+ add_token_u64(&err, dev, setup->range_length);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+
+ add_token_u8(&err, dev, OPAL_ENDLIST);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
- add_token_u8(&err, dev, OPAL_ENDLIST);
- add_token_u8(&err, dev, OPAL_ENDNAME);
- }
if (err) {
- pr_debug("Error building Setup Locking range command.\n");
+ pr_debug("Error building Setup Locking RangeStartLength command.\n");
return err;
}
@@ -1568,7 +1581,7 @@ static int setup_locking_range(struct opal_dev *dev, void *data)
static int response_get_column(const struct parsed_resp *resp,
int *iter,
- u8 column,
+ u64 column,
u64 *value)
{
const struct opal_resp_tok *tok;
@@ -1586,7 +1599,7 @@ static int response_get_column(const struct parsed_resp *resp,
n++;
if (response_get_u64(resp, n) != column) {
- pr_debug("Token %d does not match expected column %u.\n",
+ pr_debug("Token %d does not match expected column %llu.\n",
n, column);
return OPAL_INVAL_PARAM;
}
@@ -1744,6 +1757,12 @@ static int start_anybodyASP_opal_session(struct opal_dev *dev, void *data)
OPAL_ADMINSP_UID, NULL, 0);
}
+static int start_anybodyLSP_opal_session(struct opal_dev *dev, void *data)
+{
+ return start_generic_opal_session(dev, OPAL_ANYBODY_UID,
+ OPAL_LOCKINGSP_UID, NULL, 0);
+}
+
static int start_SIDASP_opal_session(struct opal_dev *dev, void *data)
{
int ret;
@@ -2285,6 +2304,74 @@ static int activate_lsp(struct opal_dev *dev, void *data)
return finalize_and_send(dev, parse_and_check_status);
}
+static int reactivate_lsp(struct opal_dev *dev, void *data)
+{
+ struct opal_lr_react *opal_react = data;
+ u8 user_lr[OPAL_UID_LENGTH];
+ int err, i;
+
+ err = cmd_start(dev, opaluid[OPAL_THISSP_UID],
+ opalmethod[OPAL_REACTIVATE]);
+
+ if (err) {
+ pr_debug("Error building Reactivate LockingSP command.\n");
+ return err;
+ }
+
+ /*
+ * If neither 'entire_table' nor 'num_lrs' is set, the device
+ * gets reactivated with SUM disabled. Only Admin1PIN will change
+ * if set.
+ */
+ if (opal_react->entire_table) {
+ /* Entire Locking table (all locking ranges) will be put in SUM. */
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u64(&err, dev, OPAL_SUM_SET_LIST);
+ add_token_bytestring(&err, dev, opaluid[OPAL_LOCKING_TABLE], OPAL_UID_LENGTH);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+ } else if (opal_react->num_lrs) {
+ /* Subset of Locking table (selected locking range(s)) to be put in SUM */
+ err = build_locking_range(user_lr, sizeof(user_lr),
+ opal_react->lr[0]);
+ if (err)
+ return err;
+
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u64(&err, dev, OPAL_SUM_SET_LIST);
+
+ add_token_u8(&err, dev, OPAL_STARTLIST);
+ add_token_bytestring(&err, dev, user_lr, OPAL_UID_LENGTH);
+ for (i = 1; i < opal_react->num_lrs; i++) {
+ user_lr[7] = opal_react->lr[i];
+ add_token_bytestring(&err, dev, user_lr, OPAL_UID_LENGTH);
+ }
+ add_token_u8(&err, dev, OPAL_ENDLIST);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+ }
+
+ /* Skipping the rangle policy parameter is same as setting its value to zero */
+ if (opal_react->range_policy && (opal_react->num_lrs || opal_react->entire_table)) {
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u64(&err, dev, OPAL_SUM_RANGE_POLICY);
+ add_token_u8(&err, dev, 1);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+ }
+
+ /*
+ * Optional parameter. If set, it changes the Admin1 PIN even when SUM
+ * is being disabled.
+ */
+ if (opal_react->new_admin_key.key_len) {
+ add_token_u8(&err, dev, OPAL_STARTNAME);
+ add_token_u64(&err, dev, OPAL_SUM_ADMIN1_PIN);
+ add_token_bytestring(&err, dev, opal_react->new_admin_key.key,
+ opal_react->new_admin_key.key_len);
+ add_token_u8(&err, dev, OPAL_ENDNAME);
+ }
+
+ return finalize_and_send(dev, parse_and_check_status);
+}
+
/* Determine if we're in the Manufactured Inactive or Active state */
static int get_lsp_lifecycle(struct opal_dev *dev, void *data)
{
@@ -2955,12 +3042,92 @@ static int opal_activate_lsp(struct opal_dev *dev,
return ret;
}
+static int opal_reactivate_lsp(struct opal_dev *dev,
+ struct opal_lr_react *opal_lr_react)
+{
+ const struct opal_step active_steps[] = {
+ { start_admin1LSP_opal_session, &opal_lr_react->key },
+ { reactivate_lsp, opal_lr_react },
+ /* No end_opal_session. The controller terminates the session */
+ };
+ int ret;
+
+ /* use either 'entire_table' parameter or set of locking ranges */
+ if (opal_lr_react->num_lrs > OPAL_MAX_LRS ||
+ (opal_lr_react->num_lrs && opal_lr_react->entire_table))
+ return -EINVAL;
+
+ ret = opal_get_key(dev, &opal_lr_react->key);
+ if (ret)
+ return ret;
+ mutex_lock(&dev->dev_lock);
+ setup_opal_dev(dev);
+ ret = execute_steps(dev, active_steps, ARRAY_SIZE(active_steps));
+ mutex_unlock(&dev->dev_lock);
+
+ return ret;
+}
+
static int opal_setup_locking_range(struct opal_dev *dev,
struct opal_user_lr_setup *opal_lrs)
{
const struct opal_step lr_steps[] = {
{ start_auth_opal_session, &opal_lrs->session },
- { setup_locking_range, opal_lrs },
+ { setup_locking_range_start_length, opal_lrs },
+ { setup_enable_range, opal_lrs },
+ { end_opal_session, }
+ }, lr_global_steps[] = {
+ { start_auth_opal_session, &opal_lrs->session },
+ { setup_enable_range, opal_lrs },
+ { end_opal_session, }
+ };
+ int ret;
+
+ ret = opal_get_key(dev, &opal_lrs->session.opal_key);
+ if (ret)
+ return ret;
+ mutex_lock(&dev->dev_lock);
+ setup_opal_dev(dev);
+ if (opal_lrs->session.opal_key.lr == 0)
+ ret = execute_steps(dev, lr_global_steps, ARRAY_SIZE(lr_global_steps));
+ else
+ ret = execute_steps(dev, lr_steps, ARRAY_SIZE(lr_steps));
+ mutex_unlock(&dev->dev_lock);
+
+ return ret;
+}
+
+static int opal_setup_locking_range_start_length(struct opal_dev *dev,
+ struct opal_user_lr_setup *opal_lrs)
+{
+ const struct opal_step lr_steps[] = {
+ { start_auth_opal_session, &opal_lrs->session },
+ { setup_locking_range_start_length, opal_lrs },
+ { end_opal_session, }
+ };
+ int ret;
+
+ /* we can not set global locking range offset or length */
+ if (opal_lrs->session.opal_key.lr == 0)
+ return -EINVAL;
+
+ ret = opal_get_key(dev, &opal_lrs->session.opal_key);
+ if (ret)
+ return ret;
+ mutex_lock(&dev->dev_lock);
+ setup_opal_dev(dev);
+ ret = execute_steps(dev, lr_steps, ARRAY_SIZE(lr_steps));
+ mutex_unlock(&dev->dev_lock);
+
+ return ret;
+}
+
+static int opal_enable_disable_range(struct opal_dev *dev,
+ struct opal_user_lr_setup *opal_lrs)
+{
+ const struct opal_step lr_steps[] = {
+ { start_auth_opal_session, &opal_lrs->session },
+ { setup_enable_range, opal_lrs },
{ end_opal_session, }
};
int ret;
@@ -3228,6 +3395,200 @@ static int opal_get_geometry(struct opal_dev *dev, void __user *data)
return 0;
}
+static int get_sum_ranges(struct opal_dev *dev, void *data)
+{
+ const char *lr_uid;
+ size_t lr_uid_len;
+ u64 val;
+ const struct opal_resp_tok *tok;
+ int err, tok_n = 2;
+ struct opal_sum_ranges *sranges = data;
+ const __u8 lr_all[OPAL_MAX_LRS] = { 0, 1, 2, 3, 4, 5, 6, 7, 8 };
+
+ err = generic_get_columns(dev, opaluid[OPAL_LOCKING_INFO_TABLE], OPAL_SUM_SET_LIST,
+ OPAL_SUM_RANGE_POLICY);
+ if (err) {
+ pr_debug("Couldn't get locking info table columns %d to %d.\n",
+ OPAL_SUM_SET_LIST, OPAL_SUM_RANGE_POLICY);
+ return err;
+ }
+
+ tok = response_get_token(&dev->parsed, tok_n);
+ if (IS_ERR(tok))
+ return PTR_ERR(tok);
+
+ if (!response_token_matches(tok, OPAL_STARTNAME)) {
+ pr_debug("Unexpected response token type %d.\n", tok_n);
+ return OPAL_INVAL_PARAM;
+ }
+ tok_n++;
+
+ if (response_get_u64(&dev->parsed, tok_n) != OPAL_SUM_SET_LIST) {
+ pr_debug("Token %d does not match expected column %u.\n",
+ tok_n, OPAL_SUM_SET_LIST);
+ return OPAL_INVAL_PARAM;
+ }
+ tok_n++;
+
+ tok = response_get_token(&dev->parsed, tok_n);
+ if (IS_ERR(tok))
+ return PTR_ERR(tok);
+
+ /*
+ * The OPAL_SUM_SET_LIST response contains two distinct values:
+ *
+ * - the list of individual locking ranges (UIDs) put in SUM. The list
+ * may also be empty signaling the SUM is disabled.
+ *
+ * - the Locking table UID if the entire Locking table is put in SUM.
+ */
+ if (response_token_matches(tok, OPAL_STARTLIST)) {
+ sranges->num_lrs = 0;
+
+ tok_n++;
+ tok = response_get_token(&dev->parsed, tok_n);
+ if (IS_ERR(tok))
+ return PTR_ERR(tok);
+
+ while (!response_token_matches(tok, OPAL_ENDLIST)) {
+ lr_uid_len = response_get_string(&dev->parsed, tok_n, &lr_uid);
+ if (lr_uid_len != OPAL_UID_LENGTH) {
+ pr_debug("Unexpected response token type %d.\n", tok_n);
+ return OPAL_INVAL_PARAM;
+ }
+
+ if (memcmp(lr_uid, opaluid[OPAL_LOCKINGRANGE_GLOBAL], OPAL_UID_LENGTH)) {
+ if (lr_uid[5] != LOCKING_RANGE_NON_GLOBAL) {
+ pr_debug("Unexpected byte %d at LR UUID position 5.\n",
+ lr_uid[5]);
+ return OPAL_INVAL_PARAM;
+ }
+ sranges->lr[sranges->num_lrs++] = lr_uid[7];
+ } else
+ sranges->lr[sranges->num_lrs++] = 0;
+
+ tok_n++;
+ tok = response_get_token(&dev->parsed, tok_n);
+ if (IS_ERR(tok))
+ return PTR_ERR(tok);
+ }
+ } else {
+ /* Only OPAL_LOCKING_TABLE UID is an alternative to OPAL_STARTLIST here. */
+ lr_uid_len = response_get_string(&dev->parsed, tok_n, &lr_uid);
+ if (lr_uid_len != OPAL_UID_LENGTH) {
+ pr_debug("Unexpected response token type %d.\n", tok_n);
+ return OPAL_INVAL_PARAM;
+ }
+
+ if (memcmp(lr_uid, opaluid[OPAL_LOCKING_TABLE], OPAL_UID_LENGTH)) {
+ pr_debug("Unexpected response UID.\n");
+ return OPAL_INVAL_PARAM;
+ }
+
+ /* sed-opal kernel API already provides following limit in Activate command */
+ sranges->num_lrs = OPAL_MAX_LRS;
+ memcpy(sranges->lr, lr_all, OPAL_MAX_LRS);
+ }
+ tok_n++;
+
+ tok = response_get_token(&dev->parsed, tok_n);
+ if (IS_ERR(tok))
+ return PTR_ERR(tok);
+
+ if (!response_token_matches(tok, OPAL_ENDNAME)) {
+ pr_debug("Unexpected response token type %d.\n", tok_n);
+ return OPAL_INVAL_PARAM;
+ }
+ tok_n++;
+
+ err = response_get_column(&dev->parsed, &tok_n, OPAL_SUM_RANGE_POLICY, &val);
+ if (err)
+ return err;
+
+ sranges->range_policy = val ? 1 : 0;
+
+ return 0;
+}
+
+static int opal_get_sum_ranges(struct opal_dev *dev, struct opal_sum_ranges *opal_sum_rngs,
+ void __user *data)
+{
+ const struct opal_step admin_steps[] = {
+ { start_admin1LSP_opal_session, &opal_sum_rngs->key },
+ { get_sum_ranges, opal_sum_rngs },
+ { end_opal_session, }
+ }, anybody_steps[] = {
+ { start_anybodyLSP_opal_session, NULL },
+ { get_sum_ranges, opal_sum_rngs },
+ { end_opal_session, }
+ };
+ int ret;
+
+ mutex_lock(&dev->dev_lock);
+ setup_opal_dev(dev);
+ if (opal_sum_rngs->key.key_len)
+ /* Use Admin1 session (authenticated by PIN) to retrieve LockingInfo columns */
+ ret = execute_steps(dev, admin_steps, ARRAY_SIZE(admin_steps));
+ else
+ /* Use Anybody session (no key) to retrieve LockingInfo columns */
+ ret = execute_steps(dev, anybody_steps, ARRAY_SIZE(anybody_steps));
+ mutex_unlock(&dev->dev_lock);
+
+ /* skip session info when copying back to uspace */
+ if (!ret && copy_to_user(data + offsetof(struct opal_sum_ranges, num_lrs),
+ (void *)opal_sum_rngs + offsetof(struct opal_sum_ranges, num_lrs),
+ sizeof(*opal_sum_rngs) - offsetof(struct opal_sum_ranges, num_lrs))) {
+ pr_debug("Error copying SUM ranges info to userspace\n");
+ return -EFAULT;
+ }
+
+ return ret;
+}
+
+static int opal_stack_reset(struct opal_dev *dev)
+{
+ struct opal_stack_reset *req;
+ struct opal_stack_reset_response *resp;
+ int ret;
+
+ mutex_lock(&dev->dev_lock);
+
+ memset(dev->cmd, 0, IO_BUFFER_LENGTH);
+ req = (struct opal_stack_reset *)dev->cmd;
+ req->extendedComID[0] = dev->comid >> 8;
+ req->extendedComID[1] = dev->comid & 0xFF;
+ req->request_code = cpu_to_be32(OPAL_STACK_RESET);
+
+ ret = dev->send_recv(dev->data, dev->comid, TCG_SECP_02,
+ dev->cmd, IO_BUFFER_LENGTH, true);
+ if (ret) {
+ pr_debug("Error sending stack reset: %d\n", ret);
+ goto out;
+ }
+
+ memset(dev->resp, 0, IO_BUFFER_LENGTH);
+ ret = dev->send_recv(dev->data, dev->comid, TCG_SECP_02,
+ dev->resp, IO_BUFFER_LENGTH, false);
+ if (ret) {
+ pr_debug("Error receiving stack reset response: %d\n", ret);
+ goto out;
+ }
+
+ resp = (struct opal_stack_reset_response *)dev->resp;
+ if (be16_to_cpu(resp->data_length) != 4) {
+ pr_debug("Stack reset pending\n");
+ ret = -EBUSY;
+ goto out;
+ }
+ if (be32_to_cpu(resp->response) != 0) {
+ pr_debug("Stack reset failed: %u\n", be32_to_cpu(resp->response));
+ ret = -EIO;
+ }
+out:
+ mutex_unlock(&dev->dev_lock);
+ return ret;
+}
+
int sed_ioctl(struct opal_dev *dev, unsigned int cmd, void __user *arg)
{
void *p;
@@ -3313,6 +3674,21 @@ int sed_ioctl(struct opal_dev *dev, unsigned int cmd, void __user *arg)
case IOC_OPAL_SET_SID_PW:
ret = opal_set_new_sid_pw(dev, p);
break;
+ case IOC_OPAL_REACTIVATE_LSP:
+ ret = opal_reactivate_lsp(dev, p);
+ break;
+ case IOC_OPAL_LR_SET_START_LEN:
+ ret = opal_setup_locking_range_start_length(dev, p);
+ break;
+ case IOC_OPAL_ENABLE_DISABLE_LR:
+ ret = opal_enable_disable_range(dev, p);
+ break;
+ case IOC_OPAL_GET_SUM_STATUS:
+ ret = opal_get_sum_ranges(dev, p, arg);
+ break;
+ case IOC_OPAL_STACK_RESET:
+ ret = opal_stack_reset(dev);
+ break;
default:
break;
diff --git a/block/t10-pi.c b/block/t10-pi.c
index d27be6041fd3..a19b4e102a83 100644
--- a/block/t10-pi.c
+++ b/block/t10-pi.c
@@ -12,462 +12,556 @@
#include <linux/unaligned.h>
#include "blk.h"
+#define APP_TAG_ESCAPE 0xffff
+#define REF_TAG_ESCAPE 0xffffffff
+
+/*
+ * This union is used for onstack allocations when the pi field is split across
+ * segments. blk_validate_integrity_limits() guarantees pi_tuple_size matches
+ * the sizeof one of these two types.
+ */
+union pi_tuple {
+ struct crc64_pi_tuple crc64_pi;
+ struct t10_pi_tuple t10_pi;
+};
+
struct blk_integrity_iter {
- void *prot_buf;
- void *data_buf;
- sector_t seed;
- unsigned int data_size;
- unsigned short interval;
- const char *disk_name;
+ struct bio *bio;
+ struct bio_integrity_payload *bip;
+ struct blk_integrity *bi;
+ struct bvec_iter data_iter;
+ struct bvec_iter prot_iter;
+ unsigned int interval_remaining;
+ u64 seed;
+ u64 csum;
};
-static __be16 t10_pi_csum(__be16 csum, void *data, unsigned int len,
- unsigned char csum_type)
+static void blk_calculate_guard(struct blk_integrity_iter *iter, void *data,
+ unsigned int len)
{
- if (csum_type == BLK_INTEGRITY_CSUM_IP)
- return (__force __be16)ip_compute_csum(data, len);
- return cpu_to_be16(crc_t10dif_update(be16_to_cpu(csum), data, len));
+ switch (iter->bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ iter->csum = crc64_nvme(iter->csum, data, len);
+ break;
+ case BLK_INTEGRITY_CSUM_CRC:
+ iter->csum = crc_t10dif_update(iter->csum, data, len);
+ break;
+ case BLK_INTEGRITY_CSUM_IP:
+ iter->csum = (__force u32)csum_partial(data, len,
+ (__force __wsum)iter->csum);
+ break;
+ default:
+ WARN_ON_ONCE(1);
+ iter->csum = U64_MAX;
+ break;
+ }
+}
+
+static void blk_integrity_csum_finish(struct blk_integrity_iter *iter)
+{
+ switch (iter->bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_IP:
+ iter->csum = (__force u16)csum_fold((__force __wsum)iter->csum);
+ break;
+ default:
+ break;
+ }
}
/*
- * Type 1 and Type 2 protection use the same format: 16 bit guard tag,
- * 16 bit app tag, 32 bit reference tag. Type 3 does not define the ref
- * tag.
+ * Update the csum for formats that have metadata padding in front of the data
+ * integrity field
*/
-static void t10_pi_generate(struct blk_integrity_iter *iter,
- struct blk_integrity *bi)
+static void blk_integrity_csum_offset(struct blk_integrity_iter *iter)
{
- u8 offset = bi->pi_offset;
- unsigned int i;
+ unsigned int offset = iter->bi->pi_offset;
+ struct bio_vec *bvec = iter->bip->bip_vec;
+
+ while (offset > 0) {
+ struct bio_vec pbv = bvec_iter_bvec(bvec, iter->prot_iter);
+ unsigned int len = min(pbv.bv_len, offset);
+ void *prot_buf = bvec_kmap_local(&pbv);
+
+ blk_calculate_guard(iter, prot_buf, len);
+ kunmap_local(prot_buf);
+ offset -= len;
+ bvec_iter_advance_single(bvec, &iter->prot_iter, len);
+ }
+ blk_integrity_csum_finish(iter);
+}
- for (i = 0 ; i < iter->data_size ; i += iter->interval) {
- struct t10_pi_tuple *pi = iter->prot_buf + offset;
+static void blk_integrity_copy_from_tuple(struct bio_integrity_payload *bip,
+ struct bvec_iter *iter, void *tuple,
+ unsigned int tuple_size)
+{
+ while (tuple_size) {
+ struct bio_vec pbv = bvec_iter_bvec(bip->bip_vec, *iter);
+ unsigned int len = min(tuple_size, pbv.bv_len);
+ void *prot_buf = bvec_kmap_local(&pbv);
+
+ memcpy(prot_buf, tuple, len);
+ kunmap_local(prot_buf);
+ bvec_iter_advance_single(bip->bip_vec, iter, len);
+ tuple_size -= len;
+ tuple += len;
+ }
+}
- pi->guard_tag = t10_pi_csum(0, iter->data_buf, iter->interval,
- bi->csum_type);
- if (offset)
- pi->guard_tag = t10_pi_csum(pi->guard_tag,
- iter->prot_buf, offset, bi->csum_type);
- pi->app_tag = 0;
+static void blk_integrity_copy_to_tuple(struct bio_integrity_payload *bip,
+ struct bvec_iter *iter, void *tuple,
+ unsigned int tuple_size)
+{
+ while (tuple_size) {
+ struct bio_vec pbv = bvec_iter_bvec(bip->bip_vec, *iter);
+ unsigned int len = min(tuple_size, pbv.bv_len);
+ void *prot_buf = bvec_kmap_local(&pbv);
+
+ memcpy(tuple, prot_buf, len);
+ kunmap_local(prot_buf);
+ bvec_iter_advance_single(bip->bip_vec, iter, len);
+ tuple_size -= len;
+ tuple += len;
+ }
+}
- if (bi->flags & BLK_INTEGRITY_REF_TAG)
- pi->ref_tag = cpu_to_be32(lower_32_bits(iter->seed));
- else
- pi->ref_tag = 0;
+static bool ext_pi_ref_escape(const u8 ref_tag[6])
+{
+ static const u8 ref_escape[6] = { 0xff, 0xff, 0xff, 0xff, 0xff, 0xff };
- iter->data_buf += iter->interval;
- iter->prot_buf += bi->metadata_size;
- iter->seed++;
- }
+ return memcmp(ref_tag, ref_escape, sizeof(ref_escape)) == 0;
}
-static blk_status_t t10_pi_verify(struct blk_integrity_iter *iter,
- struct blk_integrity *bi)
-{
- u8 offset = bi->pi_offset;
- unsigned int i;
-
- for (i = 0 ; i < iter->data_size ; i += iter->interval) {
- struct t10_pi_tuple *pi = iter->prot_buf + offset;
- __be16 csum;
-
- if (bi->flags & BLK_INTEGRITY_REF_TAG) {
- if (pi->app_tag == T10_PI_APP_ESCAPE)
- goto next;
-
- if (be32_to_cpu(pi->ref_tag) !=
- lower_32_bits(iter->seed)) {
- pr_err("%s: ref tag error at location %llu " \
- "(rcvd %u)\n", iter->disk_name,
- (unsigned long long)
- iter->seed, be32_to_cpu(pi->ref_tag));
- return BLK_STS_PROTECTION;
- }
- } else {
- if (pi->app_tag == T10_PI_APP_ESCAPE &&
- pi->ref_tag == T10_PI_REF_ESCAPE)
- goto next;
+static blk_status_t blk_verify_ext_pi(struct blk_integrity_iter *iter,
+ struct crc64_pi_tuple *pi)
+{
+ u64 seed = lower_48_bits(iter->seed);
+ u64 guard = get_unaligned_be64(&pi->guard_tag);
+ u64 ref = get_unaligned_be48(pi->ref_tag);
+ u16 app = get_unaligned_be16(&pi->app_tag);
+
+ if (iter->bi->flags & BLK_INTEGRITY_REF_TAG) {
+ if (app == APP_TAG_ESCAPE)
+ return BLK_STS_OK;
+ if (ref != seed) {
+ pr_err("%s: ref tag error at location %llu (rcvd %llu)\n",
+ iter->bio->bi_bdev->bd_disk->disk_name, seed,
+ ref);
+ return BLK_STS_PROTECTION;
}
+ } else if (app == APP_TAG_ESCAPE && ext_pi_ref_escape(pi->ref_tag)) {
+ return BLK_STS_OK;
+ }
+
+ if (guard != iter->csum) {
+ pr_err("%s: guard tag error at sector %llu (rcvd %016llx, want %016llx)\n",
+ iter->bio->bi_bdev->bd_disk->disk_name, iter->seed,
+ guard, iter->csum);
+ return BLK_STS_PROTECTION;
+ }
+
+ return BLK_STS_OK;
+}
- csum = t10_pi_csum(0, iter->data_buf, iter->interval,
- bi->csum_type);
- if (offset)
- csum = t10_pi_csum(csum, iter->prot_buf, offset,
- bi->csum_type);
-
- if (pi->guard_tag != csum) {
- pr_err("%s: guard tag error at sector %llu " \
- "(rcvd %04x, want %04x)\n", iter->disk_name,
- (unsigned long long)iter->seed,
- be16_to_cpu(pi->guard_tag), be16_to_cpu(csum));
+static blk_status_t blk_verify_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi, u16 guard)
+{
+ u32 seed = lower_32_bits(iter->seed);
+ u32 ref = get_unaligned_be32(&pi->ref_tag);
+ u16 app = get_unaligned_be16(&pi->app_tag);
+
+ if (iter->bi->flags & BLK_INTEGRITY_REF_TAG) {
+ if (app == APP_TAG_ESCAPE)
+ return BLK_STS_OK;
+ if (ref != seed) {
+ pr_err("%s: ref tag error at location %u (rcvd %u)\n",
+ iter->bio->bi_bdev->bd_disk->disk_name, seed,
+ ref);
return BLK_STS_PROTECTION;
}
+ } else if (app == APP_TAG_ESCAPE && ref == REF_TAG_ESCAPE) {
+ return BLK_STS_OK;
+ }
-next:
- iter->data_buf += iter->interval;
- iter->prot_buf += bi->metadata_size;
- iter->seed++;
+ if (guard != (u16)iter->csum) {
+ pr_err("%s: guard tag error at sector %llu (rcvd %04x, want %04x)\n",
+ iter->bio->bi_bdev->bd_disk->disk_name, iter->seed,
+ guard, (u16)iter->csum);
+ return BLK_STS_PROTECTION;
}
return BLK_STS_OK;
}
-/**
- * t10_pi_type1_prepare - prepare PI prior submitting request to device
- * @rq: request with PI that should be prepared
- *
- * For Type 1/Type 2, the virtual start sector is the one that was
- * originally submitted by the block layer for the ref_tag usage. Due to
- * partitioning, MD/DM cloning, etc. the actual physical start sector is
- * likely to be different. Remap protection information to match the
- * physical LBA.
- */
-static void t10_pi_type1_prepare(struct request *rq)
+static blk_status_t blk_verify_t10_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi)
{
- struct blk_integrity *bi = &rq->q->limits.integrity;
- const int tuple_sz = bi->metadata_size;
- u32 ref_tag = t10_pi_ref_tag(rq);
- u8 offset = bi->pi_offset;
- struct bio *bio;
+ u16 guard = get_unaligned_be16(&pi->guard_tag);
- __rq_for_each_bio(bio, rq) {
- struct bio_integrity_payload *bip = bio_integrity(bio);
- u32 virt = bip_get_seed(bip) & 0xffffffff;
- struct bio_vec iv;
- struct bvec_iter iter;
+ return blk_verify_pi(iter, pi, guard);
+}
- /* Already remapped? */
- if (bip->bip_flags & BIP_MAPPED_INTEGRITY)
- break;
+static blk_status_t blk_verify_ip_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi)
+{
+ u16 guard = get_unaligned((u16 *)&pi->guard_tag);
- bip_for_each_vec(iv, bip, iter) {
- unsigned int j;
- void *p;
-
- p = bvec_kmap_local(&iv);
- for (j = 0; j < iv.bv_len; j += tuple_sz) {
- struct t10_pi_tuple *pi = p + offset;
-
- if (be32_to_cpu(pi->ref_tag) == virt)
- pi->ref_tag = cpu_to_be32(ref_tag);
- virt++;
- ref_tag++;
- p += tuple_sz;
- }
- kunmap_local(p);
- }
+ return blk_verify_pi(iter, pi, guard);
+}
- bip->bip_flags |= BIP_MAPPED_INTEGRITY;
+static blk_status_t blk_integrity_verify(struct blk_integrity_iter *iter,
+ union pi_tuple *tuple)
+{
+ switch (iter->bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ return blk_verify_ext_pi(iter, &tuple->crc64_pi);
+ case BLK_INTEGRITY_CSUM_CRC:
+ return blk_verify_t10_pi(iter, &tuple->t10_pi);
+ case BLK_INTEGRITY_CSUM_IP:
+ return blk_verify_ip_pi(iter, &tuple->t10_pi);
+ default:
+ return BLK_STS_OK;
}
}
-/**
- * t10_pi_type1_complete - prepare PI prior returning request to the blk layer
- * @rq: request with PI that should be prepared
- * @nr_bytes: total bytes to prepare
- *
- * For Type 1/Type 2, the virtual start sector is the one that was
- * originally submitted by the block layer for the ref_tag usage. Due to
- * partitioning, MD/DM cloning, etc. the actual physical start sector is
- * likely to be different. Since the physical start sector was submitted
- * to the device, we should remap it back to virtual values expected by the
- * block layer.
- */
-static void t10_pi_type1_complete(struct request *rq, unsigned int nr_bytes)
+static void blk_set_ext_pi(struct blk_integrity_iter *iter,
+ struct crc64_pi_tuple *pi)
{
- struct blk_integrity *bi = &rq->q->limits.integrity;
- unsigned intervals = nr_bytes >> bi->interval_exp;
- const int tuple_sz = bi->metadata_size;
- u32 ref_tag = t10_pi_ref_tag(rq);
- u8 offset = bi->pi_offset;
- struct bio *bio;
+ put_unaligned_be64(iter->csum, &pi->guard_tag);
+ put_unaligned_be16(0, &pi->app_tag);
+ put_unaligned_be48(iter->seed, &pi->ref_tag);
+}
- __rq_for_each_bio(bio, rq) {
- struct bio_integrity_payload *bip = bio_integrity(bio);
- u32 virt = bip_get_seed(bip) & 0xffffffff;
- struct bio_vec iv;
- struct bvec_iter iter;
-
- bip_for_each_vec(iv, bip, iter) {
- unsigned int j;
- void *p;
-
- p = bvec_kmap_local(&iv);
- for (j = 0; j < iv.bv_len && intervals; j += tuple_sz) {
- struct t10_pi_tuple *pi = p + offset;
-
- if (be32_to_cpu(pi->ref_tag) == ref_tag)
- pi->ref_tag = cpu_to_be32(virt);
- virt++;
- ref_tag++;
- intervals--;
- p += tuple_sz;
- }
- kunmap_local(p);
- }
- }
+static void blk_set_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi, __be16 csum)
+{
+ put_unaligned(csum, &pi->guard_tag);
+ put_unaligned_be16(0, &pi->app_tag);
+ put_unaligned_be32(iter->seed, &pi->ref_tag);
}
-static __be64 ext_pi_crc64(u64 crc, void *data, unsigned int len)
+static void blk_set_t10_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi)
{
- return cpu_to_be64(crc64_nvme(crc, data, len));
+ blk_set_pi(iter, pi, cpu_to_be16((u16)iter->csum));
}
-static void ext_pi_crc64_generate(struct blk_integrity_iter *iter,
- struct blk_integrity *bi)
+static void blk_set_ip_pi(struct blk_integrity_iter *iter,
+ struct t10_pi_tuple *pi)
{
- u8 offset = bi->pi_offset;
- unsigned int i;
+ blk_set_pi(iter, pi, (__force __be16)(u16)iter->csum);
+}
- for (i = 0 ; i < iter->data_size ; i += iter->interval) {
- struct crc64_pi_tuple *pi = iter->prot_buf + offset;
+static void blk_integrity_set(struct blk_integrity_iter *iter,
+ union pi_tuple *tuple)
+{
+ switch (iter->bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ return blk_set_ext_pi(iter, &tuple->crc64_pi);
+ case BLK_INTEGRITY_CSUM_CRC:
+ return blk_set_t10_pi(iter, &tuple->t10_pi);
+ case BLK_INTEGRITY_CSUM_IP:
+ return blk_set_ip_pi(iter, &tuple->t10_pi);
+ default:
+ WARN_ON_ONCE(1);
+ return;
+ }
+}
- pi->guard_tag = ext_pi_crc64(0, iter->data_buf, iter->interval);
- if (offset)
- pi->guard_tag = ext_pi_crc64(be64_to_cpu(pi->guard_tag),
- iter->prot_buf, offset);
- pi->app_tag = 0;
+static blk_status_t blk_integrity_interval(struct blk_integrity_iter *iter,
+ bool verify)
+{
+ blk_status_t ret = BLK_STS_OK;
+ union pi_tuple tuple;
+ void *ptuple = &tuple;
+ struct bio_vec pbv;
+
+ blk_integrity_csum_offset(iter);
+ pbv = bvec_iter_bvec(iter->bip->bip_vec, iter->prot_iter);
+ if (pbv.bv_len >= iter->bi->pi_tuple_size) {
+ ptuple = bvec_kmap_local(&pbv);
+ bvec_iter_advance_single(iter->bip->bip_vec, &iter->prot_iter,
+ iter->bi->metadata_size - iter->bi->pi_offset);
+ } else if (verify) {
+ blk_integrity_copy_to_tuple(iter->bip, &iter->prot_iter,
+ ptuple, iter->bi->pi_tuple_size);
+ }
- if (bi->flags & BLK_INTEGRITY_REF_TAG)
- put_unaligned_be48(iter->seed, pi->ref_tag);
- else
- put_unaligned_be48(0ULL, pi->ref_tag);
+ if (verify)
+ ret = blk_integrity_verify(iter, ptuple);
+ else
+ blk_integrity_set(iter, ptuple);
- iter->data_buf += iter->interval;
- iter->prot_buf += bi->metadata_size;
- iter->seed++;
+ if (likely(ptuple != &tuple)) {
+ kunmap_local(ptuple);
+ } else if (!verify) {
+ blk_integrity_copy_from_tuple(iter->bip, &iter->prot_iter,
+ ptuple, iter->bi->pi_tuple_size);
}
+
+ iter->interval_remaining = 1 << iter->bi->interval_exp;
+ iter->csum = 0;
+ iter->seed++;
+ return ret;
}
-static bool ext_pi_ref_escape(const u8 ref_tag[6])
+static blk_status_t blk_integrity_iterate(struct bio *bio,
+ struct bvec_iter *data_iter,
+ bool verify)
{
- static const u8 ref_escape[6] = { 0xff, 0xff, 0xff, 0xff, 0xff, 0xff };
+ struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk);
+ struct bio_integrity_payload *bip = bio_integrity(bio);
+ struct blk_integrity_iter iter = {
+ .bio = bio,
+ .bip = bip,
+ .bi = bi,
+ .data_iter = *data_iter,
+ .prot_iter = bip->bip_iter,
+ .interval_remaining = 1 << bi->interval_exp,
+ .seed = data_iter->bi_sector,
+ .csum = 0,
+ };
+ blk_status_t ret = BLK_STS_OK;
+
+ while (iter.data_iter.bi_size && ret == BLK_STS_OK) {
+ struct bio_vec bv = bvec_iter_bvec(iter.bio->bi_io_vec,
+ iter.data_iter);
+ void *kaddr = bvec_kmap_local(&bv);
+ void *data = kaddr;
+ unsigned int len;
+
+ bvec_iter_advance_single(iter.bio->bi_io_vec, &iter.data_iter,
+ bv.bv_len);
+ while (bv.bv_len && ret == BLK_STS_OK) {
+ len = min(iter.interval_remaining, bv.bv_len);
+ blk_calculate_guard(&iter, data, len);
+ bv.bv_len -= len;
+ data += len;
+ iter.interval_remaining -= len;
+ if (!iter.interval_remaining)
+ ret = blk_integrity_interval(&iter, verify);
+ }
+ kunmap_local(kaddr);
+ }
- return memcmp(ref_tag, ref_escape, sizeof(ref_escape)) == 0;
+ return ret;
}
-static blk_status_t ext_pi_crc64_verify(struct blk_integrity_iter *iter,
- struct blk_integrity *bi)
-{
- u8 offset = bi->pi_offset;
- unsigned int i;
-
- for (i = 0; i < iter->data_size; i += iter->interval) {
- struct crc64_pi_tuple *pi = iter->prot_buf + offset;
- u64 ref, seed;
- __be64 csum;
-
- if (bi->flags & BLK_INTEGRITY_REF_TAG) {
- if (pi->app_tag == T10_PI_APP_ESCAPE)
- goto next;
-
- ref = get_unaligned_be48(pi->ref_tag);
- seed = lower_48_bits(iter->seed);
- if (ref != seed) {
- pr_err("%s: ref tag error at location %llu (rcvd %llu)\n",
- iter->disk_name, seed, ref);
- return BLK_STS_PROTECTION;
- }
- } else {
- if (pi->app_tag == T10_PI_APP_ESCAPE &&
- ext_pi_ref_escape(pi->ref_tag))
- goto next;
- }
+void bio_integrity_generate(struct bio *bio)
+{
+ struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk);
- csum = ext_pi_crc64(0, iter->data_buf, iter->interval);
- if (offset)
- csum = ext_pi_crc64(be64_to_cpu(csum), iter->prot_buf,
- offset);
+ switch (bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ case BLK_INTEGRITY_CSUM_CRC:
+ case BLK_INTEGRITY_CSUM_IP:
+ blk_integrity_iterate(bio, &bio->bi_iter, false);
+ break;
+ default:
+ break;
+ }
+}
- if (pi->guard_tag != csum) {
- pr_err("%s: guard tag error at sector %llu " \
- "(rcvd %016llx, want %016llx)\n",
- iter->disk_name, (unsigned long long)iter->seed,
- be64_to_cpu(pi->guard_tag), be64_to_cpu(csum));
- return BLK_STS_PROTECTION;
- }
+blk_status_t bio_integrity_verify(struct bio *bio, struct bvec_iter *saved_iter)
+{
+ struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk);
-next:
- iter->data_buf += iter->interval;
- iter->prot_buf += bi->metadata_size;
- iter->seed++;
+ switch (bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ case BLK_INTEGRITY_CSUM_CRC:
+ case BLK_INTEGRITY_CSUM_IP:
+ return blk_integrity_iterate(bio, saved_iter, true);
+ default:
+ break;
}
return BLK_STS_OK;
}
-static void ext_pi_type1_prepare(struct request *rq)
+/*
+ * Advance @iter past the protection offset for protection formats that
+ * contain front padding on the metadata region.
+ */
+static void blk_pi_advance_offset(struct blk_integrity *bi,
+ struct bio_integrity_payload *bip,
+ struct bvec_iter *iter)
{
- struct blk_integrity *bi = &rq->q->limits.integrity;
- const int tuple_sz = bi->metadata_size;
- u64 ref_tag = ext_pi_ref_tag(rq);
- u8 offset = bi->pi_offset;
- struct bio *bio;
+ unsigned int offset = bi->pi_offset;
- __rq_for_each_bio(bio, rq) {
- struct bio_integrity_payload *bip = bio_integrity(bio);
- u64 virt = lower_48_bits(bip_get_seed(bip));
- struct bio_vec iv;
- struct bvec_iter iter;
+ while (offset > 0) {
+ struct bio_vec bv = mp_bvec_iter_bvec(bip->bip_vec, *iter);
+ unsigned int len = min(bv.bv_len, offset);
- /* Already remapped? */
- if (bip->bip_flags & BIP_MAPPED_INTEGRITY)
- break;
+ bvec_iter_advance_single(bip->bip_vec, iter, len);
+ offset -= len;
+ }
+}
- bip_for_each_vec(iv, bip, iter) {
- unsigned int j;
- void *p;
-
- p = bvec_kmap_local(&iv);
- for (j = 0; j < iv.bv_len; j += tuple_sz) {
- struct crc64_pi_tuple *pi = p + offset;
- u64 ref = get_unaligned_be48(pi->ref_tag);
-
- if (ref == virt)
- put_unaligned_be48(ref_tag, pi->ref_tag);
- virt++;
- ref_tag++;
- p += tuple_sz;
- }
- kunmap_local(p);
- }
+static void *blk_tuple_remap_begin(union pi_tuple *tuple,
+ struct blk_integrity *bi,
+ struct bio_integrity_payload *bip,
+ struct bvec_iter *iter)
+{
+ struct bvec_iter titer;
+ struct bio_vec pbv;
- bip->bip_flags |= BIP_MAPPED_INTEGRITY;
+ blk_pi_advance_offset(bi, bip, iter);
+ pbv = bvec_iter_bvec(bip->bip_vec, *iter);
+ if (likely(pbv.bv_len >= bi->pi_tuple_size))
+ return bvec_kmap_local(&pbv);
+
+ /*
+ * We need to preserve the state of the original iter for the
+ * copy_from_tuple at the end, so make a temp iter for here.
+ */
+ titer = *iter;
+ blk_integrity_copy_to_tuple(bip, &titer, tuple, bi->pi_tuple_size);
+ return tuple;
+}
+
+static void blk_tuple_remap_end(union pi_tuple *tuple, void *ptuple,
+ struct blk_integrity *bi,
+ struct bio_integrity_payload *bip,
+ struct bvec_iter *iter)
+{
+ unsigned int len = bi->metadata_size - bi->pi_offset;
+
+ if (likely(ptuple != tuple)) {
+ kunmap_local(ptuple);
+ } else {
+ blk_integrity_copy_from_tuple(bip, iter, ptuple,
+ bi->pi_tuple_size);
+ len -= bi->pi_tuple_size;
}
+
+ bvec_iter_advance(bip->bip_vec, iter, len);
}
-static void ext_pi_type1_complete(struct request *rq, unsigned int nr_bytes)
+static void blk_set_ext_unmap_ref(struct crc64_pi_tuple *pi, u64 virt,
+ u64 ref_tag)
{
- struct blk_integrity *bi = &rq->q->limits.integrity;
- unsigned intervals = nr_bytes >> bi->interval_exp;
- const int tuple_sz = bi->metadata_size;
- u64 ref_tag = ext_pi_ref_tag(rq);
- u8 offset = bi->pi_offset;
- struct bio *bio;
+ u64 ref = get_unaligned_be48(&pi->ref_tag);
- __rq_for_each_bio(bio, rq) {
- struct bio_integrity_payload *bip = bio_integrity(bio);
- u64 virt = lower_48_bits(bip_get_seed(bip));
- struct bio_vec iv;
- struct bvec_iter iter;
-
- bip_for_each_vec(iv, bip, iter) {
- unsigned int j;
- void *p;
-
- p = bvec_kmap_local(&iv);
- for (j = 0; j < iv.bv_len && intervals; j += tuple_sz) {
- struct crc64_pi_tuple *pi = p + offset;
- u64 ref = get_unaligned_be48(pi->ref_tag);
-
- if (ref == ref_tag)
- put_unaligned_be48(virt, pi->ref_tag);
- virt++;
- ref_tag++;
- intervals--;
- p += tuple_sz;
- }
- kunmap_local(p);
- }
+ if (ref == lower_48_bits(ref_tag) && ref != lower_48_bits(virt))
+ put_unaligned_be48(virt, pi->ref_tag);
+}
+
+static void blk_set_t10_unmap_ref(struct t10_pi_tuple *pi, u32 virt,
+ u32 ref_tag)
+{
+ u32 ref = get_unaligned_be32(&pi->ref_tag);
+
+ if (ref == ref_tag && ref != virt)
+ put_unaligned_be32(virt, &pi->ref_tag);
+}
+
+static void blk_reftag_remap_complete(struct blk_integrity *bi,
+ union pi_tuple *tuple, u64 virt, u64 ref)
+{
+ switch (bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ blk_set_ext_unmap_ref(&tuple->crc64_pi, virt, ref);
+ break;
+ case BLK_INTEGRITY_CSUM_CRC:
+ case BLK_INTEGRITY_CSUM_IP:
+ blk_set_t10_unmap_ref(&tuple->t10_pi, virt, ref);
+ break;
+ default:
+ WARN_ON_ONCE(1);
+ break;
}
}
-void bio_integrity_generate(struct bio *bio)
+static void blk_set_ext_map_ref(struct crc64_pi_tuple *pi, u64 virt,
+ u64 ref_tag)
{
- struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk);
- struct bio_integrity_payload *bip = bio_integrity(bio);
- struct blk_integrity_iter iter;
- struct bvec_iter bviter;
- struct bio_vec bv;
-
- iter.disk_name = bio->bi_bdev->bd_disk->disk_name;
- iter.interval = 1 << bi->interval_exp;
- iter.seed = bio->bi_iter.bi_sector;
- iter.prot_buf = bvec_virt(bip->bip_vec);
- bio_for_each_segment(bv, bio, bviter) {
- void *kaddr = bvec_kmap_local(&bv);
+ u64 ref = get_unaligned_be48(&pi->ref_tag);
- iter.data_buf = kaddr;
- iter.data_size = bv.bv_len;
- switch (bi->csum_type) {
- case BLK_INTEGRITY_CSUM_CRC64:
- ext_pi_crc64_generate(&iter, bi);
- break;
- case BLK_INTEGRITY_CSUM_CRC:
- case BLK_INTEGRITY_CSUM_IP:
- t10_pi_generate(&iter, bi);
- break;
- default:
- break;
- }
- kunmap_local(kaddr);
+ if (ref == lower_48_bits(virt) && ref != ref_tag)
+ put_unaligned_be48(ref_tag, pi->ref_tag);
+}
+
+static void blk_set_t10_map_ref(struct t10_pi_tuple *pi, u32 virt, u32 ref_tag)
+{
+ u32 ref = get_unaligned_be32(&pi->ref_tag);
+
+ if (ref == virt && ref != ref_tag)
+ put_unaligned_be32(ref_tag, &pi->ref_tag);
+}
+
+static void blk_reftag_remap_prepare(struct blk_integrity *bi,
+ union pi_tuple *tuple,
+ u64 virt, u64 ref)
+{
+ switch (bi->csum_type) {
+ case BLK_INTEGRITY_CSUM_CRC64:
+ blk_set_ext_map_ref(&tuple->crc64_pi, virt, ref);
+ break;
+ case BLK_INTEGRITY_CSUM_CRC:
+ case BLK_INTEGRITY_CSUM_IP:
+ blk_set_t10_map_ref(&tuple->t10_pi, virt, ref);
+ break;
+ default:
+ WARN_ON_ONCE(1);
+ break;
}
}
-blk_status_t bio_integrity_verify(struct bio *bio, struct bvec_iter *saved_iter)
+static void __blk_reftag_remap(struct bio *bio, struct blk_integrity *bi,
+ unsigned *intervals, u64 *ref, bool prep)
{
- struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk);
struct bio_integrity_payload *bip = bio_integrity(bio);
- struct blk_integrity_iter iter;
- struct bvec_iter bviter;
- struct bio_vec bv;
+ struct bvec_iter iter = bip->bip_iter;
+ u64 virt = bip_get_seed(bip);
+ union pi_tuple *ptuple;
+ union pi_tuple tuple;
- /*
- * At the moment verify is called bi_iter has been advanced during split
- * and completion, so use the copy created during submission here.
- */
- iter.disk_name = bio->bi_bdev->bd_disk->disk_name;
- iter.interval = 1 << bi->interval_exp;
- iter.seed = saved_iter->bi_sector;
- iter.prot_buf = bvec_virt(bip->bip_vec);
- __bio_for_each_segment(bv, bio, bviter, *saved_iter) {
- void *kaddr = bvec_kmap_local(&bv);
- blk_status_t ret = BLK_STS_OK;
+ if (prep && bip->bip_flags & BIP_MAPPED_INTEGRITY) {
+ *ref += bio->bi_iter.bi_size >> bi->interval_exp;
+ return;
+ }
- iter.data_buf = kaddr;
- iter.data_size = bv.bv_len;
- switch (bi->csum_type) {
- case BLK_INTEGRITY_CSUM_CRC64:
- ret = ext_pi_crc64_verify(&iter, bi);
- break;
- case BLK_INTEGRITY_CSUM_CRC:
- case BLK_INTEGRITY_CSUM_IP:
- ret = t10_pi_verify(&iter, bi);
- break;
- default:
- break;
- }
- kunmap_local(kaddr);
+ while (iter.bi_size && *intervals) {
+ ptuple = blk_tuple_remap_begin(&tuple, bi, bip, &iter);
+
+ if (prep)
+ blk_reftag_remap_prepare(bi, ptuple, virt, *ref);
+ else
+ blk_reftag_remap_complete(bi, ptuple, virt, *ref);
- if (ret)
- return ret;
+ blk_tuple_remap_end(&tuple, ptuple, bi, bip, &iter);
+ (*intervals)--;
+ (*ref)++;
+ virt++;
}
- return BLK_STS_OK;
+ if (prep)
+ bip->bip_flags |= BIP_MAPPED_INTEGRITY;
}
-void blk_integrity_prepare(struct request *rq)
+static void blk_integrity_remap(struct request *rq, unsigned int nr_bytes,
+ bool prep)
{
struct blk_integrity *bi = &rq->q->limits.integrity;
+ u64 ref = blk_rq_pos(rq) >> (bi->interval_exp - SECTOR_SHIFT);
+ unsigned intervals = nr_bytes >> bi->interval_exp;
+ struct bio *bio;
if (!(bi->flags & BLK_INTEGRITY_REF_TAG))
return;
- if (bi->csum_type == BLK_INTEGRITY_CSUM_CRC64)
- ext_pi_type1_prepare(rq);
- else
- t10_pi_type1_prepare(rq);
+ __rq_for_each_bio(bio, rq) {
+ __blk_reftag_remap(bio, bi, &intervals, &ref, prep);
+ if (!intervals)
+ break;
+ }
}
-void blk_integrity_complete(struct request *rq, unsigned int nr_bytes)
+void blk_integrity_prepare(struct request *rq)
{
- struct blk_integrity *bi = &rq->q->limits.integrity;
-
- if (!(bi->flags & BLK_INTEGRITY_REF_TAG))
- return;
+ blk_integrity_remap(rq, blk_rq_bytes(rq), true);
+}
- if (bi->csum_type == BLK_INTEGRITY_CSUM_CRC64)
- ext_pi_type1_complete(rq, nr_bytes);
- else
- t10_pi_type1_complete(rq, nr_bytes);
+void blk_integrity_complete(struct request *rq, unsigned int nr_bytes)
+{
+ blk_integrity_remap(rq, nr_bytes, false);
}