diff options
| author | Mark Brown <broonie@kernel.org> | 2026-08-21 13:40:43 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-08-21 13:40:43 +0100 |
| commit | 25f3c66839f7020a703af0c9becc64cae0269e89 (patch) | |
| tree | e02f0fd79f18f24aa9053a85a92859b58292903f | |
| parent | effdff7168b0216825821d0e85172a1cb8f7a70e (diff) | |
| parent | eef628bce8e18ce1eb8531c396bf55e36f86c62c (diff) | |
| download | linux-next-25f3c66839f7020a703af0c9becc64cae0269e89.tar.gz linux-next-25f3c66839f7020a703af0c9becc64cae0269e89.zip | |
Merge branch 'mm-nonmm-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm
86 files changed, 2589 insertions, 988 deletions
diff --git a/.get_maintainer.ignore b/.get_maintainer.ignore index 60b6b2a374cd..5ad082b4dd03 100644 --- a/.get_maintainer.ignore +++ b/.get_maintainer.ignore @@ -5,4 +5,5 @@ Askar Safin <safinaskar@gmail.com> Christoph Hellwig <hch@lst.de> Jeff Kirsher <jeffrey.t.kirsher@intel.com> Marc Gonzalez <marc.w.gonzalez@free.fr> +Nathan Chancellor <nathan@kernel.org> Ralf Baechle <ralf@linux-mips.org> @@ -193,6 +193,7 @@ Carlos Bilbao <carlos.bilbao@kernel.org> <bilbao@vt.edu> Changbin Du <changbin.du@intel.com> <changbin.du@gmail.com> Chao Yu <chao@kernel.org> <chao2.yu@samsung.com> Chao Yu <chao@kernel.org> <yuchao0@huawei.com> +Charlie Jenkins <thecharlesjenkins@gmail.com> <charlie@rivosinc.com> Chen-Yu Tsai <wens@kernel.org> <wens@csie.org> Chester Lin <chester62515@gmail.com> <clin@suse.com> Chris Chiu <chris.chiu@canonical.com> <chiu@endlessm.com> @@ -528,6 +529,8 @@ Leo Yan <leo.yan@linux.dev> <leo.yan@linaro.org> Liam R. Howlett <liam@infradead.org> <Liam.Howlett@oracle.com> Liam Mark <quic_lmark@quicinc.com> <lmark@codeaurora.org> Linas Vepstas <linas@austin.ibm.com> +Linfeng Sun <linfeng.sun.dev@gmail.com> <slf@hdu.edu.cn> +Linfeng Sun <linfeng.sun.dev@gmail.com> <linfeng.sun.dev@gamil.com> Linus Lüssing <linus.luessing@c0d3.blue> <linus.luessing@ascom.ch> Linus Lüssing <linus.luessing@c0d3.blue> <linus.luessing@web.de> Linus Lüssing <linus.luessing@c0d3.blue> <ll@simonwunderlich.de> @@ -876,7 +879,8 @@ Szymon Wilczek <swilczek.lx@gmail.com> <szymonwilczek@gmx.com> Takashi YOSHII <takashi.yoshii.zj@renesas.com> Tamir Duberstein <tamird@kernel.org> <tamird@gmail.com> Tamizh Chelvam Raja <quic_tamizhr@quicinc.com> <tamizhr@codeaurora.org> -Taniya Das <quic_tdas@quicinc.com> <tdas@codeaurora.org> +Taniya Das <taniya.das@oss.qualcomm.com> <quic_tdas@quicinc.com> +Taniya Das <taniya.das@oss.qualcomm.com> <tdas@codeaurora.org> Tanzir Hasan <tanzhasanwork@gmail.com> <tanzirh@google.com> Tejun Heo <htejun@gmail.com> Tomeu Vizoso <tomeu@tomeuvizoso.net> <tomeu.vizoso@collabora.com> diff --git a/MAINTAINERS b/MAINTAINERS index 115f6ca6151d..51336bb51367 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -15449,7 +15449,9 @@ M: Yang Xu <xuyang2018.jy@fujitsu.com> M: Xiao Yang <yangx.jy@fujitsu.com> L: ltp@lists.linux.it (subscribers-only) S: Maintained +C: irc://irc.libera.chat/ltp W: https://linux-test-project.readthedocs.io/ +Q: https://patchwork.kernel.org/project/ltp/list/ T: git https://github.com/linux-test-project/ltp.git LTR390 AMBIENT/UV LIGHT SENSOR DRIVER diff --git a/drivers/gpu/drm/drm_modeset_lock.c b/drivers/gpu/drm/drm_modeset_lock.c index 2c806b0146d6..e14814c30d8c 100644 --- a/drivers/gpu/drm/drm_modeset_lock.c +++ b/drivers/gpu/drm/drm_modeset_lock.c @@ -149,11 +149,7 @@ void drm_modeset_lock_all(struct drm_device *dev) int ret; ctx = kzalloc_obj(*ctx, GFP_KERNEL | __GFP_NOFAIL); - if (WARN_ON(!ctx)) - return; - mutex_lock(&config->mutex); - drm_modeset_acquire_init(ctx, 0); retry: diff --git a/drivers/infiniband/hw/cxgb4/mem.c b/drivers/infiniband/hw/cxgb4/mem.c index cd1b01014198..08631de17c0a 100644 --- a/drivers/infiniband/hw/cxgb4/mem.c +++ b/drivers/infiniband/hw/cxgb4/mem.c @@ -74,11 +74,8 @@ static int _c4iw_write_mem_dma_aligned(struct c4iw_rdev *rdev, u32 addr, c4iw_init_wr_wait(wr_waitp); wr_len = roundup(sizeof(*req) + sizeof(*sgl), 16); - if (!skb) { + if (!skb) skb = alloc_skb(wr_len, GFP_KERNEL | __GFP_NOFAIL); - if (!skb) - return -ENOMEM; - } set_wr_txq(skb, CPL_PRIORITY_CONTROL, 0); req = __skb_put_zero(skb, wr_len); @@ -134,11 +131,8 @@ static int _c4iw_write_mem_inline(struct c4iw_rdev *rdev, u32 addr, u32 len, roundup(copy_len, T4_ULPTX_MIN_IO), 16); - if (!skb) { + if (!skb) skb = alloc_skb(wr_len, GFP_KERNEL | __GFP_NOFAIL); - if (!skb) - return -ENOMEM; - } set_wr_txq(skb, CPL_PRIORITY_CONTROL, 0); req = __skb_put_zero(skb, wr_len); diff --git a/drivers/pps/clients/pps-gpio.c b/drivers/pps/clients/pps-gpio.c index 402f910f3e25..73ec2c7335e5 100644 --- a/drivers/pps/clients/pps-gpio.c +++ b/drivers/pps/clients/pps-gpio.c @@ -31,38 +31,37 @@ struct pps_gpio_device_data { struct gpio_desc *echo_pin; struct timer_list echo_timer; /* timer to reset echo active state */ bool assert_falling_edge; - bool capture_clear; unsigned int echo_active_ms; /* PPS echo active duration */ unsigned long echo_timeout; /* timer timeout value in jiffies */ + struct pps_event_time ts; /* timestamp captured in hardirq */ }; /* * Report the PPS event */ -static irqreturn_t pps_gpio_irq_handler(int irq, void *data) +/* + * Primary hardirq handler -- runs in hardirq context even on PREEMPT_RT. + * Only captures the timestamp; all other work is deferred to the thread. + */ +static irqreturn_t pps_gpio_irq_hardirq(int irq, void *data) { - const struct pps_gpio_device_data *info; - struct pps_event_time ts; - int rising_edge; - - /* Get the time stamp first */ - pps_get_ts(&ts); - - info = data; - - /* Small trick to bypass the check on edge's direction when capture_clear is unset */ - rising_edge = info->capture_clear ? - gpiod_get_value(info->gpio_pin) : !info->assert_falling_edge; - if ((rising_edge && !info->assert_falling_edge) || - (!rising_edge && info->assert_falling_edge)) - pps_event(info->pps, &ts, PPS_CAPTUREASSERT, data); - else if (info->capture_clear && - ((rising_edge && info->assert_falling_edge) || - (!rising_edge && !info->assert_falling_edge))) - pps_event(info->pps, &ts, PPS_CAPTURECLEAR, data); - else - dev_warn_ratelimited(&info->pps->dev, "IRQ did not trigger any PPS event\n"); + struct pps_gpio_device_data *info = data; + + pps_get_ts(&info->ts); + + return IRQ_WAKE_THREAD; +} + +/* + * Threaded handler -- processes the PPS event using the timestamp + * captured in hardirq context above. + */ +static irqreturn_t pps_gpio_irq_thread(int irq, void *data) +{ + struct pps_gpio_device_data *info = data; + + pps_event(info->pps, &info->ts, PPS_CAPTUREASSERT, data); return IRQ_HANDLED; } @@ -78,11 +77,6 @@ static void pps_gpio_echo(struct pps_device *pps, int event, void *data) if (pps->params.mode & PPS_ECHOASSERT) gpiod_set_value(info->echo_pin, 1); break; - - case PPS_CAPTURECLEAR: - if (pps->params.mode & PPS_ECHOCLEAR) - gpiod_set_value(info->echo_pin, 1); - break; } /* fire the timer */ @@ -144,15 +138,8 @@ static int pps_gpio_setup(struct device *dev) static unsigned long get_irqf_trigger_flags(const struct pps_gpio_device_data *data) { - unsigned long flags = data->assert_falling_edge ? - IRQF_TRIGGER_FALLING : IRQF_TRIGGER_RISING; - - if (data->capture_clear) { - flags |= ((flags & IRQF_TRIGGER_RISING) ? - IRQF_TRIGGER_FALLING : IRQF_TRIGGER_RISING); - } - - return flags; + return data->assert_falling_edge ? IRQF_TRIGGER_FALLING : + IRQF_TRIGGER_RISING; } static int pps_gpio_probe(struct platform_device *pdev) @@ -185,9 +172,6 @@ static int pps_gpio_probe(struct platform_device *pdev) /* initialize PPS specific parts of the bookkeeping data structure. */ data->info.mode = PPS_CAPTUREASSERT | PPS_OFFSETASSERT | PPS_ECHOASSERT | PPS_CANWAIT | PPS_TSFMT_TSPEC; - if (data->capture_clear) - data->info.mode |= PPS_CAPTURECLEAR | PPS_OFFSETCLEAR | - PPS_ECHOCLEAR; data->info.owner = THIS_MODULE; snprintf(data->info.name, PPS_MAX_NAME_LEN - 1, "%s.%d", pdev->name, pdev->id); @@ -199,8 +183,6 @@ static int pps_gpio_probe(struct platform_device *pdev) /* register PPS source */ pps_default_params = PPS_CAPTUREASSERT | PPS_OFFSETASSERT; - if (data->capture_clear) - pps_default_params |= PPS_CAPTURECLEAR | PPS_OFFSETCLEAR; data->pps = pps_register_source(&data->info, pps_default_params); if (IS_ERR(data->pps)) { dev_err(dev, "failed to register IRQ %d as PPS source\n", @@ -209,8 +191,10 @@ static int pps_gpio_probe(struct platform_device *pdev) } /* register IRQ interrupt handler */ - ret = request_irq(data->irq, pps_gpio_irq_handler, - get_irqf_trigger_flags(data), data->info.name, data); + ret = request_threaded_irq(data->irq, + pps_gpio_irq_hardirq, pps_gpio_irq_thread, + get_irqf_trigger_flags(data) | IRQF_ONESHOT, + data->info.name, data); if (ret) { pps_unregister_source(data->pps); dev_err(dev, "failed to acquire IRQ %d\n", data->irq); diff --git a/drivers/pps/kc.c b/drivers/pps/kc.c index fbd23295afd7..4f8fffa7edd6 100644 --- a/drivers/pps/kc.c +++ b/drivers/pps/kc.c @@ -38,6 +38,14 @@ int pps_kc_bind(struct pps_device *pps, struct pps_bind_args *bind_args) /* Check if another consumer is already bound */ spin_lock_irq(&pps_kc_hardpps_lock); + /* + * Don't allow PPS_KC_BIND on a removed device. + */ + if (pps->kc_removed) { + spin_unlock_irq(&pps_kc_hardpps_lock); + return -ENODEV; + } + if (bind_args->edge == 0) if (pps_kc_hardpps_dev == pps) { pps_kc_hardpps_mode = 0; @@ -79,6 +87,8 @@ int pps_kc_bind(struct pps_device *pps, struct pps_bind_args *bind_args) void pps_kc_remove(struct pps_device *pps) { spin_lock_irq(&pps_kc_hardpps_lock); + + pps->kc_removed = true; if (pps == pps_kc_hardpps_dev) { pps_kc_hardpps_mode = 0; pps_kc_hardpps_dev = NULL; diff --git a/drivers/pps/pps.c b/drivers/pps/pps.c index de1122bb69ea..7ecdd774a44b 100644 --- a/drivers/pps/pps.c +++ b/drivers/pps/pps.c @@ -66,13 +66,19 @@ static int pps_cdev_pps_fetch(struct pps_device *pps, struct pps_fdata *fdata) err = wait_event_interruptible(pps->queue, ev != pps->last_ev); else { + struct timespec64 ts; unsigned long ticks; dev_dbg(&pps->dev, "timeout %lld.%09d\n", (long long) fdata->timeout.sec, fdata->timeout.nsec); - ticks = fdata->timeout.sec * HZ; - ticks += fdata->timeout.nsec / (NSEC_PER_SEC / HZ); + + if (fdata->timeout.sec < 0) + return -ETIMEDOUT; + + ts.tv_sec = fdata->timeout.sec; + ts.tv_nsec = fdata->timeout.nsec; + ticks = timespec64_to_jiffies(&ts); if (ticks != 0) { err = wait_event_interruptible_timeout( diff --git a/drivers/rapidio/devices/rio_mport_cdev.c b/drivers/rapidio/devices/rio_mport_cdev.c index 009b3b595bbf..ad82c2108a56 100644 --- a/drivers/rapidio/devices/rio_mport_cdev.c +++ b/drivers/rapidio/devices/rio_mport_cdev.c @@ -564,9 +564,13 @@ static void dma_req_free(struct kref *ref) } if (req->map) { - mutex_lock(&req->map->md->buf_mutex); - kref_put(&req->map->ref, mport_release_mapping); - mutex_unlock(&req->map->md->buf_mutex); + struct rio_mport_mapping *map = req->map; + struct mport_dev *md = map->md; + + mutex_lock(&md->buf_mutex); + req->map = NULL; + kref_put(&map->ref, mport_release_mapping); + mutex_unlock(&md->buf_mutex); } kref_put(&priv->dma_ref, mport_release_dma); diff --git a/drivers/rapidio/rio-scan.c b/drivers/rapidio/rio-scan.c index dcd6619a4b02..3cc25d053451 100644 --- a/drivers/rapidio/rio-scan.c +++ b/drivers/rapidio/rio-scan.c @@ -874,6 +874,7 @@ static struct rio_net *rio_scan_alloc_net(struct rio_mport *mport, net->dev.release = rio_scan_release_dev; if (rio_add_net(net)) { put_device(&net->dev); + mport->net = NULL; net = NULL; } } diff --git a/fs/fat/inode.c b/fs/fat/inode.c index 5ea6f74a2a3f..f775a004cae1 100644 --- a/fs/fat/inode.c +++ b/fs/fat/inode.c @@ -1508,8 +1508,9 @@ static int fat_read_static_bpb(struct super_block *sb, int error = -EINVAL; unsigned i; - /* 16-bit DOS 1.x reliably wrote bootstrap short-jmp code */ - if (b->ignored[0] != 0xeb || b->ignored[2] != 0x90) { + /* 16-bit DOS 1.x reliably wrote bootstrap short-jmp or near-jmp code */ + if ((b->ignored[0] != 0xeb || b->ignored[2] != 0x90) && + (b->ignored[0] != 0xe9)) { if (!silent) fat_msg(sb, KERN_ERR, "%s; no bootstrapping code", notdos1x); diff --git a/fs/fat/misc.c b/fs/fat/misc.c index be18f6b5819b..e79762cf1975 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -133,7 +133,11 @@ int fat_chain_add(struct inode *inode, int new_dclus, int nr_cluster) ret = fat_ent_read(inode, &fatent, last); if (ret >= 0) { int wait = inode_needs_sync(inode); + int old = ret; + ret = fat_ent_write(inode, &fatent, new_dclus, wait); + if (ret < 0) + fat_ent_write(inode, &fatent, old, wait); fatent_brelse(&fatent); } if (ret < 0) diff --git a/fs/fat/nfs.c b/fs/fat/nfs.c index 509eea96a457..6e1b371711ed 100644 --- a/fs/fat/nfs.c +++ b/fs/fat/nfs.c @@ -250,8 +250,10 @@ struct inode *fat_rebuild_parent(struct super_block *sb, int parent_logstart) MSDOS_I(dummy_grand_parent)->i_pos = -1; } - if (!fat_scan_logstart(dummy_grand_parent, clus_to_match, &sinfo)) + if (!fat_scan_logstart(dummy_grand_parent, clus_to_match, &sinfo)) { parent = fat_build_inode(sb, sinfo.de, sinfo.i_pos); + brelse(sinfo.bh); + } brelse(parent_bh); iput(dummy_grand_parent); diff --git a/fs/ocfs2/acl.c b/fs/ocfs2/acl.c index af1e2cedb217..090ec60fb576 100644 --- a/fs/ocfs2/acl.c +++ b/fs/ocfs2/acl.c @@ -110,8 +110,7 @@ static void *ocfs2_acl_to_xattr(const struct posix_acl *acl, size_t *size) return ocfs2_acl; } -static struct posix_acl *ocfs2_get_acl_nolock(struct inode *inode, - int type, +static struct posix_acl *ocfs2_get_acl_nolock(struct inode *inode, int type, struct buffer_head *di_bh) { int name_index; @@ -349,63 +348,105 @@ int ocfs2_acl_chmod(struct inode *inode, struct buffer_head *bh) * Initialize the ACLs of a new inode. If parent directory has default ACL, * then clone to new inode. Called from ocfs2_mknod. */ -int ocfs2_init_acl(handle_t *handle, - struct inode *inode, - struct inode *dir, - struct buffer_head *di_bh, - struct buffer_head *dir_bh, - struct ocfs2_alloc_context *meta_ac, - struct ocfs2_alloc_context *data_ac) +void ocfs2_acl_init_release(struct ocfs2_acl_state *state) +{ + posix_acl_release(state->default_acl); + posix_acl_release(state->acl); + state->default_acl = NULL; + state->acl = NULL; +} + +int ocfs2_acl_init_prepare(struct inode *inode, struct inode *dir, + struct buffer_head *dir_bh, + struct ocfs2_acl_state *state) { struct ocfs2_super *osb = OCFS2_SB(inode->i_sb); - struct posix_acl *acl = NULL; - int ret = 0, ret2; - umode_t mode; - - if (!S_ISLNK(inode->i_mode)) { - if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { - down_read(&OCFS2_I(dir)->ip_xattr_sem); - acl = ocfs2_get_acl_nolock(dir, ACL_TYPE_DEFAULT, - dir_bh); - up_read(&OCFS2_I(dir)->ip_xattr_sem); - if (IS_ERR(acl)) - return PTR_ERR(acl); + int ret = 0; + + state->default_acl = NULL; + state->acl = NULL; + state->mode = inode->i_mode; + + if (S_ISLNK(inode->i_mode)) + return 0; + + if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { + down_read(&OCFS2_I(dir)->ip_xattr_sem); + state->default_acl = + ocfs2_get_acl_nolock(dir, ACL_TYPE_DEFAULT, dir_bh); + up_read(&OCFS2_I(dir)->ip_xattr_sem); + if (IS_ERR(state->default_acl)) { + ret = PTR_ERR(state->default_acl); + state->default_acl = NULL; + return ret; } - if (!acl) { - mode = inode->i_mode & ~current_umask(); - ret = ocfs2_acl_set_mode(inode, di_bh, handle, mode); - if (ret) { - mlog_errno(ret); + if (state->default_acl) { + state->acl = posix_acl_dup(state->default_acl); + if (!state->acl) { + ret = -ENOMEM; goto cleanup; } + ret = __posix_acl_create(&state->acl, GFP_NOFS, + &state->mode); + if (ret < 0) + goto cleanup; + if (ret == 0) { + posix_acl_release(state->acl); + state->acl = NULL; + } + if (!S_ISDIR(inode->i_mode)) { + posix_acl_release(state->default_acl); + state->default_acl = NULL; + } + } else { + state->mode &= ~current_umask(); } + } else { + state->mode &= ~current_umask(); } - if ((osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) && acl) { - if (S_ISDIR(inode->i_mode)) { + + return 0; +cleanup: + ocfs2_acl_init_release(state); + return ret; +} + +int ocfs2_init_acl(handle_t *handle, struct inode *inode, + struct buffer_head *di_bh, + struct ocfs2_alloc_context *meta_ac, + struct ocfs2_alloc_context *data_ac, + struct ocfs2_acl_state *state) +{ + struct ocfs2_super *osb = OCFS2_SB(inode->i_sb); + int ret = 0; + + if (S_ISLNK(inode->i_mode)) + return 0; + + if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { + if (S_ISDIR(inode->i_mode) && state->default_acl) { ret = ocfs2_set_acl(handle, inode, di_bh, - ACL_TYPE_DEFAULT, acl, - meta_ac, data_ac); + ACL_TYPE_DEFAULT, + state->default_acl, meta_ac, + data_ac); if (ret) - goto cleanup; + return ret; } - mode = inode->i_mode; - ret = __posix_acl_create(&acl, GFP_NOFS, &mode); - if (ret < 0) - return ret; + } - ret2 = ocfs2_acl_set_mode(inode, di_bh, handle, mode); - if (ret2) { - mlog_errno(ret2); - ret = ret2; - goto cleanup; - } - if (ret > 0) { - ret = ocfs2_set_acl(handle, inode, - di_bh, ACL_TYPE_ACCESS, - acl, meta_ac, data_ac); + ret = ocfs2_acl_set_mode(inode, di_bh, handle, state->mode); + if (ret) { + mlog_errno(ret); + return ret; + } + + if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { + if (state->acl) { + ret = ocfs2_set_acl(handle, inode, di_bh, + ACL_TYPE_ACCESS, state->acl, + meta_ac, data_ac); } } -cleanup: - posix_acl_release(acl); + return ret; } diff --git a/fs/ocfs2/acl.h b/fs/ocfs2/acl.h index 667c6f03fa60..a91f9ce278d6 100644 --- a/fs/ocfs2/acl.h +++ b/fs/ocfs2/acl.h @@ -20,9 +20,20 @@ struct posix_acl *ocfs2_iop_get_acl(struct inode *inode, int type, bool rcu); int ocfs2_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ocfs2_acl_chmod(struct inode *, struct buffer_head *); -extern int ocfs2_init_acl(handle_t *, struct inode *, struct inode *, - struct buffer_head *, struct buffer_head *, - struct ocfs2_alloc_context *, - struct ocfs2_alloc_context *); +struct ocfs2_acl_state { + struct posix_acl *default_acl; + struct posix_acl *acl; + umode_t mode; +}; + +int ocfs2_acl_init_prepare(struct inode *inode, struct inode *dir, + struct buffer_head *dir_bh, + struct ocfs2_acl_state *state); +void ocfs2_acl_init_release(struct ocfs2_acl_state *state); +int ocfs2_init_acl(handle_t *handle, struct inode *inode, + struct buffer_head *di_bh, + struct ocfs2_alloc_context *meta_ac, + struct ocfs2_alloc_context *data_ac, + struct ocfs2_acl_state *state); #endif /* OCFS2_ACL_H */ diff --git a/fs/ocfs2/cluster/heartbeat.c b/fs/ocfs2/cluster/heartbeat.c index d12784aaaa4b..1c3def99bb07 100644 --- a/fs/ocfs2/cluster/heartbeat.c +++ b/fs/ocfs2/cluster/heartbeat.c @@ -15,6 +15,7 @@ #include <linux/file.h> #include <linux/kthread.h> #include <linux/configfs.h> +#include <linux/mutex.h> #include <linux/random.h> #include <linux/crc32.h> #include <linux/time.h> @@ -43,6 +44,14 @@ static DECLARE_RWSEM(o2hb_callback_sem); * whenever any of the threads sees activity from the node in its region. */ static DEFINE_SPINLOCK(o2hb_live_lock); +/* + * Serializes region pin/unpin dependency management (o2hb_dependent_users + * and the o2nm_depend_item()/o2nm_undepend_item() calls). o2hb_region_pin() + * has to drop o2hb_live_lock across the sleeping o2nm_depend_item(), so the + * spinlock alone can no longer keep pin and unpin mutually exclusive; this + * mutex, taken outside o2hb_live_lock, does. + */ +static DEFINE_MUTEX(o2hb_dependency_mutex); static struct list_head o2hb_live_slots[O2NM_MAX_NODES]; static unsigned long o2hb_live_node_bitmap[BITS_TO_LONGS(O2NM_MAX_NODES)]; static LIST_HEAD(o2hb_node_events); @@ -138,7 +147,7 @@ static unsigned int o2hb_dependent_users; * In global heartbeat mode, we pin/unpin all o2hb regions. This solution * works for both file system and userdlm domains. */ -static int o2hb_region_pin(const char *region_uuid); +static int o2hb_region_pin(const char *region_uuid, bool from_callback); static void o2hb_region_unpin(const char *region_uuid); /* Only sets a new threshold if there are no active regions. @@ -203,6 +212,7 @@ struct o2hb_region { /* protected by the hr_callback_sem */ struct task_struct *hr_task; + u8 hr_node_num; unsigned int hr_blocks; unsigned long long hr_start_block; @@ -257,6 +267,9 @@ struct o2hb_region { /* Message key for negotiate timeout message. */ unsigned int hr_key; struct list_head hr_handler_list; + /* Serializes timeout arming against failed-start and teardown. */ + struct mutex hr_arming_mutex; + bool hr_stopping; /* last hb status, 0 for success, other value for error. */ int hr_last_hb_status; @@ -271,6 +284,9 @@ struct o2hb_bio_wait_ctxt { atomic_t wc_num_reqs; struct completion wc_io_complete; int wc_error; + /* On-stack bio used by the synchronous write path only. */ + struct bio wc_write_bio; + struct bio_vec wc_write_bvec; }; #define O2HB_NEGO_TIMEOUT_MS (O2HB_MAX_WRITE_TIMEOUT_MS/2) @@ -321,9 +337,14 @@ static void o2hb_write_timeout(struct work_struct *work) static void o2hb_arm_timeout(struct o2hb_region *reg) { + mutex_lock(®->hr_arming_mutex); + + if (reg->hr_stopping) + goto out_unlock; + /* Arm writeout only after thread reaches steady state */ if (atomic_read(®->hr_steady_iterations) != 0) - return; + goto out_unlock; mlog(ML_HEARTBEAT, "Queue write timeout for %u ms\n", O2HB_MAX_WRITE_TIMEOUT_MS); @@ -342,6 +363,18 @@ static void o2hb_arm_timeout(struct o2hb_region *reg) schedule_delayed_work(®->hr_nego_timeout_work, msecs_to_jiffies(O2HB_NEGO_TIMEOUT_MS)); bitmap_zero(reg->hr_nego_node_bitmap, O2NM_MAX_NODES); + +out_unlock: + mutex_unlock(®->hr_arming_mutex); +} + +static void o2hb_queue_nego_timeout(struct o2hb_region *reg, + unsigned long delay) +{ + mutex_lock(®->hr_arming_mutex); + if (!reg->hr_stopping) + schedule_delayed_work(®->hr_nego_timeout_work, delay); + mutex_unlock(®->hr_arming_mutex); } static void o2hb_disarm_timeout(struct o2hb_region *reg) @@ -350,12 +383,25 @@ static void o2hb_disarm_timeout(struct o2hb_region *reg) cancel_delayed_work_sync(®->hr_nego_timeout_work); } -static int o2hb_send_nego_msg(int key, int type, u8 target) +static void o2hb_set_region_stopping(struct o2hb_region *reg, bool stopping) +{ + mutex_lock(®->hr_arming_mutex); + reg->hr_stopping = stopping; + mutex_unlock(®->hr_arming_mutex); +} + +static void o2hb_quiesce_timeout(struct o2hb_region *reg) +{ + o2hb_set_region_stopping(reg, true); + o2hb_disarm_timeout(reg); +} + +static int o2hb_send_nego_msg(int key, int type, u8 target, u8 node_num) { struct o2hb_nego_msg msg; int status, ret; - msg.node_num = o2nm_this_node(); + msg.node_num = node_num; again: ret = o2net_send_message(type, key, &msg, sizeof(msg), target, &status); @@ -373,8 +419,10 @@ static void o2hb_nego_timeout(struct work_struct *work) unsigned long live_node_bitmap[BITS_TO_LONGS(O2NM_MAX_NODES)]; int master_node, i, ret; struct o2hb_region *reg; + u8 node_num; reg = container_of(work, struct o2hb_region, hr_nego_timeout_work.work); + node_num = reg->hr_node_num; /* don't negotiate timeout if last hb failed since it is very * possible io failed. Should let write timeout fence self. */ @@ -385,10 +433,10 @@ static void o2hb_nego_timeout(struct work_struct *work) /* lowest node as master node to make negotiate decision. */ master_node = find_first_bit(live_node_bitmap, O2NM_MAX_NODES); - if (master_node == o2nm_this_node()) { + if (master_node == node_num) { if (!test_bit(master_node, reg->hr_nego_node_bitmap)) { printk(KERN_NOTICE "o2hb: node %d hb write hung for %ds on region %s (%pg).\n", - o2nm_this_node(), O2HB_NEGO_TIMEOUT_MS/1000, + node_num, O2HB_NEGO_TIMEOUT_MS / 1000, config_item_name(®->hr_item), reg_bdev(reg)); set_bit(master_node, reg->hr_nego_node_bitmap); } @@ -397,8 +445,7 @@ static void o2hb_nego_timeout(struct work_struct *work) /* check negotiate bitmap every second to do timeout * approve decision. */ - schedule_delayed_work(®->hr_nego_timeout_work, - msecs_to_jiffies(1000)); + o2hb_queue_nego_timeout(reg, msecs_to_jiffies(1000)); return; } @@ -417,7 +464,7 @@ static void o2hb_nego_timeout(struct work_struct *work) mlog(ML_HEARTBEAT, "send NEGO_APPROVE msg to node %d\n", i); ret = o2hb_send_nego_msg(reg->hr_key, - O2HB_NEGO_APPROVE_MSG, i); + O2HB_NEGO_APPROVE_MSG, i, node_num); if (ret) mlog(ML_ERROR, "send NEGO_APPROVE msg to node %d fail %d\n", i, ret); @@ -425,10 +472,10 @@ static void o2hb_nego_timeout(struct work_struct *work) } else { /* negotiate timeout with master node. */ printk(KERN_NOTICE "o2hb: node %d hb write hung for %ds on region %s (%pg), negotiate timeout with node %d.\n", - o2nm_this_node(), O2HB_NEGO_TIMEOUT_MS/1000, config_item_name(®->hr_item), + node_num, O2HB_NEGO_TIMEOUT_MS / 1000, config_item_name(®->hr_item), reg_bdev(reg), master_node); ret = o2hb_send_nego_msg(reg->hr_key, O2HB_NEGO_TIMEOUT_MSG, - master_node); + master_node, node_num); if (ret) mlog(ML_ERROR, "send NEGO_TIMEOUT msg to node %d fail %d\n", master_node, ret); @@ -504,6 +551,23 @@ static void o2hb_bio_end_io(struct bio *bio) bio_put(bio); } +/* + * End I/O for the synchronous write path. The write bio is embedded in + * the wait ctxt rather than allocated, so it must not be freed here; it + * is torn down with bio_uninit() once the caller has waited on it. + */ +static void o2hb_write_bio_end_io(struct bio *bio) +{ + struct o2hb_bio_wait_ctxt *wc = bio->bi_private; + + if (bio->bi_status) { + mlog(ML_ERROR, "IO Error %d\n", bio->bi_status); + wc->wc_error = blk_status_to_errno(bio->bi_status); + } + + o2hb_bio_wait_dec(wc, 1); +} + /* Setup a Bio to cover I/O against num_slots slots starting at * start_slot. */ static struct bio *o2hb_setup_one_bio(struct o2hb_region *reg, @@ -519,16 +583,12 @@ static struct bio *o2hb_setup_one_bio(struct o2hb_region *reg, struct bio *bio; struct page *page; - /* Testing has shown this allocation to take long enough under - * GFP_KERNEL that the local node can get fenced. It would be - * nicest if we could pre-allocate these bios and avoid this - * all together. */ - bio = bio_alloc(reg_bdev(reg), 16, opf, GFP_ATOMIC); - if (!bio) { - mlog(ML_ERROR, "Could not alloc slots BIO!\n"); - bio = ERR_PTR(-ENOMEM); - goto bail; - } + /* + * The heartbeat runs in process context and can sleep, so use + * GFP_NOFS. It is backed by the fs_bio_set mempool and thus cannot + * fail, while avoiding recursion back into the filesystem. + */ + bio = bio_alloc(reg_bdev(reg), 16, opf, GFP_NOFS); /* Must put everything in 512 byte sectors for the bio... */ bio->bi_iter.bi_sector = (reg->hr_start_block + cs) << (bits - 9); @@ -553,7 +613,6 @@ static struct bio *o2hb_setup_one_bio(struct o2hb_region *reg, vec_start = 0; } -bail: *current_slot = cs; return bio; } @@ -563,7 +622,6 @@ static int o2hb_read_slots(struct o2hb_region *reg, unsigned int max_slots) { unsigned int current_slot = begin_slot; - int status; struct o2hb_bio_wait_ctxt wc; struct bio *bio; @@ -572,51 +630,51 @@ static int o2hb_read_slots(struct o2hb_region *reg, while(current_slot < max_slots) { bio = o2hb_setup_one_bio(reg, &wc, ¤t_slot, max_slots, REQ_OP_READ); - if (IS_ERR(bio)) { - status = PTR_ERR(bio); - mlog_errno(status); - goto bail_and_wait; - } - atomic_inc(&wc.wc_num_reqs); submit_bio(bio); } - status = 0; - -bail_and_wait: o2hb_wait_on_io(&wc); - if (wc.wc_error && !status) - status = wc.wc_error; - return status; + return wc.wc_error; } static int o2hb_issue_node_write(struct o2hb_region *reg, struct o2hb_bio_wait_ctxt *write_wc) { - int status; unsigned int slot; - struct bio *bio; + unsigned int bits = reg->hr_block_bits; + unsigned int spp = reg->hr_slots_per_page; + unsigned int vec_start, vec_len; + struct page *page; + struct bio *bio = &write_wc->wc_write_bio; o2hb_bio_wait_init(write_wc); - slot = o2nm_this_node(); + slot = reg->hr_node_num; + if (slot >= O2NM_MAX_NODES) + return -EINVAL; - bio = o2hb_setup_one_bio(reg, write_wc, &slot, slot+1, - REQ_OP_WRITE | REQ_SYNC); - if (IS_ERR(bio)) { - status = PTR_ERR(bio); - mlog_errno(status); - goto bail; - } + /* + * The heartbeat write always covers our own single slot, i.e. one + * block that lives within a single page. Use an on-stack bio (embedded + * in write_wc) so this fence-critical path never has to allocate. + */ + bio_init(bio, reg_bdev(reg), &write_wc->wc_write_bvec, 1, + REQ_OP_WRITE | REQ_SYNC); + bio->bi_iter.bi_sector = (reg->hr_start_block + slot) << (bits - 9); + bio->bi_private = write_wc; + bio->bi_end_io = o2hb_write_bio_end_io; + + page = reg->hr_slot_data[slot / spp]; + vec_start = (slot << bits) % PAGE_SIZE; + vec_len = PAGE_SIZE / spp; + __bio_add_page(bio, page, vec_len, vec_start); atomic_inc(&write_wc->wc_num_reqs); submit_bio(bio); - status = 0; -bail: - return status; + return 0; } static u32 o2hb_compute_block_crc_le(struct o2hb_region *reg, @@ -670,8 +728,12 @@ static int o2hb_check_own_slot(struct o2hb_region *reg) struct o2hb_disk_slot *slot; struct o2hb_disk_heartbeat_block *hb_block; char *errstr; + u8 node_num = reg->hr_node_num; - slot = ®->hr_slots[o2nm_this_node()]; + if (node_num >= O2NM_MAX_NODES) + return 0; + + slot = ®->hr_slots[node_num]; /* Don't check on our 1st timestamp */ if (!slot->ds_last_time) return 0; @@ -712,7 +774,10 @@ static inline void o2hb_prepare_block(struct o2hb_region *reg, struct o2hb_disk_slot *slot; struct o2hb_disk_heartbeat_block *hb_block; - node_num = o2nm_this_node(); + node_num = reg->hr_node_num; + if (node_num >= O2NM_MAX_NODES) + return; + slot = ®->hr_slots[node_num]; hb_block = (struct o2hb_disk_heartbeat_block *)slot->ds_raw_block; @@ -1146,6 +1211,7 @@ static int o2hb_do_disk_heartbeat(struct o2hb_region *reg) * people we find in our steady state have seen us. */ o2hb_wait_on_io(&write_wc); + bio_uninit(&write_wc.wc_write_bio); if (write_wc.wc_error) { /* Do not re-arm the write timeout on I/O error - we * can't be sure that the new block ever made it to @@ -1206,7 +1272,7 @@ static int o2hb_thread(void *data) set_user_nice(current, MIN_NICE); /* Pin node */ - ret = o2nm_depend_this_node(); + ret = o2nm_depend_node(reg->hr_node_num); if (ret) { mlog(ML_ERROR, "Node has been deleted, ret = %d\n", ret); reg->hr_node_deleted = 1; @@ -1215,7 +1281,8 @@ static int o2hb_thread(void *data) } while (!kthread_should_stop() && - !reg->hr_unclean_stop && !reg->hr_aborted_start) { + !reg->hr_unclean_stop && !reg->hr_aborted_start && + o2nm_this_node() == reg->hr_node_num) { /* We track the time spent inside * o2hb_do_disk_heartbeat so that we avoid more than * hr_timeout_ms between disk writes. On busy systems @@ -1257,14 +1324,16 @@ static int o2hb_thread(void *data) if (!reg->hr_unclean_stop && !reg->hr_aborted_start) { o2hb_prepare_block(reg, 0); ret = o2hb_issue_node_write(reg, &write_wc); - if (ret == 0) + if (ret == 0) { o2hb_wait_on_io(&write_wc); - else + bio_uninit(&write_wc.wc_write_bio); + } else { mlog_errno(ret); + } } /* Unpin node */ - o2nm_undepend_this_node(); + o2nm_undepend_node(reg->hr_node_num); mlog(ML_HEARTBEAT|ML_KTHREAD, "o2hb thread exiting\n"); @@ -1456,13 +1525,38 @@ void o2hb_init(void) o2hb_debug_init(); } -/* if we're already in a callback then we're already serialized by the sem */ -static void o2hb_fill_node_map_from_callback(unsigned long *map, - unsigned int bits) +static void __o2hb_fill_node_map(unsigned long *map, unsigned int bits) { bitmap_copy(map, o2hb_live_node_bitmap, bits); } +void o2hb_callback_read_lock(void) +{ + down_read(&o2hb_callback_sem); +} + +void o2hb_callback_read_unlock(void) +{ + up_read(&o2hb_callback_sem); +} + +void o2hb_synchronize_callbacks(void) +{ + down_write(&o2hb_callback_sem); + up_write(&o2hb_callback_sem); +} + +/* + * Callers must already hold o2hb_callback_sem for read or write so the copy + * stays serialized with callback delivery. + */ +void o2hb_fill_node_map_locked(unsigned long *map, unsigned int bits) +{ + spin_lock(&o2hb_live_lock); + __o2hb_fill_node_map(map, bits); + spin_unlock(&o2hb_live_lock); +} + /* * get a map of all nodes that are heartbeating in any regions */ @@ -1470,11 +1564,9 @@ void o2hb_fill_node_map(unsigned long *map, unsigned int bits) { /* callers want to serialize this map and callbacks so that they * can trust that they don't miss nodes coming to the party */ - down_read(&o2hb_callback_sem); - spin_lock(&o2hb_live_lock); - o2hb_fill_node_map_from_callback(map, bits); - spin_unlock(&o2hb_live_lock); - up_read(&o2hb_callback_sem); + o2hb_callback_read_lock(); + o2hb_fill_node_map_locked(map, bits); + o2hb_callback_read_unlock(); } EXPORT_SYMBOL_GPL(o2hb_fill_node_map); @@ -1522,6 +1614,8 @@ static void o2hb_region_release(struct config_item *item) mlog(ML_HEARTBEAT, "hb region release (%pg)\n", reg_bdev(reg)); + o2hb_quiesce_timeout(reg); + o2net_unregister_and_flush_handler_list(®->hr_handler_list); o2hb_unmap_slot_data(reg); if (reg->hr_bdev_file) @@ -1537,7 +1631,6 @@ static void o2hb_region_release(struct config_item *item) list_del(®->hr_all_item); spin_unlock(&o2hb_live_lock); - o2net_unregister_handler_list(®->hr_handler_list); kfree(reg); } @@ -1791,7 +1884,8 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, /* We can't heartbeat without having had our node number * configured yet. */ - if (o2nm_this_node() == O2NM_MAX_NODES) + reg->hr_node_num = o2nm_this_node(); + if (reg->hr_node_num == O2NM_MAX_NODES) return -EINVAL; ret = kstrtol(p, 0, &fd); @@ -1851,9 +1945,6 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, goto out; } - INIT_DELAYED_WORK(®->hr_write_timeout_work, o2hb_write_timeout); - INIT_DELAYED_WORK(®->hr_nego_timeout_work, o2hb_nego_timeout); - /* * A node is considered live after it has beat LIVE_THRESHOLD * times. We're not steady until we've given them a chance @@ -1873,6 +1964,7 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, atomic_set(®->hr_steady_iterations, live_threshold); /* unsteady_iterations is triple the steady_iterations */ atomic_set(®->hr_unsteady_iterations, (live_threshold * 3)); + o2hb_set_region_stopping(reg, false); hb_task = kthread_run(o2hb_thread, reg, "o2hb-%s", reg->hr_item.ci_name); @@ -1922,6 +2014,8 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, out: if (ret < 0) { + o2hb_quiesce_timeout(reg); + spin_lock(&o2hb_live_lock); hb_task = reg->hr_task; reg->hr_task = NULL; @@ -2036,6 +2130,7 @@ static struct config_item *o2hb_heartbeat_group_make_item(struct config_group *g ret = -ENAMETOOLONG; goto free; } + reg->hr_node_num = O2NM_MAX_NODES; spin_lock(&o2hb_live_lock); reg->hr_region_num = 0; @@ -2060,6 +2155,10 @@ static struct config_item *o2hb_heartbeat_group_make_item(struct config_group *g */ reg->hr_key = crc32_le(reg->hr_region_num + O2NM_MAX_REGIONS, name, strlen(name)); + mutex_init(®->hr_arming_mutex); + reg->hr_stopping = true; + INIT_DELAYED_WORK(®->hr_write_timeout_work, o2hb_write_timeout); + INIT_DELAYED_WORK(®->hr_nego_timeout_work, o2hb_nego_timeout); INIT_LIST_HEAD(®->hr_handler_list); ret = o2net_register_handler(O2HB_NEGO_TIMEOUT_MSG, reg->hr_key, sizeof(struct o2hb_nego_msg), @@ -2080,7 +2179,7 @@ static struct config_item *o2hb_heartbeat_group_make_item(struct config_group *g return ®->hr_item; unregister_handler: - o2net_unregister_handler_list(®->hr_handler_list); + o2net_unregister_and_flush_handler_list(®->hr_handler_list); remove_item: spin_lock(&o2hb_live_lock); list_del(®->hr_all_item); @@ -2099,6 +2198,8 @@ static void o2hb_heartbeat_group_drop_item(struct config_group *group, struct o2hb_region *reg = to_o2hb_region(item); int quorum_region = 0; + o2hb_quiesce_timeout(reg); + /* stop the thread when the user removes the region dir */ spin_lock(&o2hb_live_lock); hb_task = reg->hr_task; @@ -2142,6 +2243,7 @@ static void o2hb_heartbeat_group_drop_item(struct config_group *group, * If global heartbeat active and there are dependent users, * pin all regions if quorum region count <= CUT_OFF */ + mutex_lock(&o2hb_dependency_mutex); spin_lock(&o2hb_live_lock); if (!o2hb_dependent_users) @@ -2149,10 +2251,11 @@ static void o2hb_heartbeat_group_drop_item(struct config_group *group, if (bitmap_weight(o2hb_quorum_region_bitmap, O2NM_MAX_REGIONS) <= O2HB_PIN_CUT_OFF) - o2hb_region_pin(NULL); + o2hb_region_pin(NULL, true); unlock: spin_unlock(&o2hb_live_lock); + mutex_unlock(&o2hb_dependency_mutex); } static ssize_t o2hb_heartbeat_group_dead_threshold_show(struct config_item *item, @@ -2290,48 +2393,113 @@ EXPORT_SYMBOL_GPL(o2hb_setup_callback); * In local, we only pin the matching region. In global we pin all the active * regions. */ -static int o2hb_region_pin(const char *region_uuid) +static int o2hb_region_pin(const char *region_uuid, bool from_callback) { - int ret = 0, found = 0; - struct o2hb_region *reg; + int ret = 0, found; + struct o2hb_region *reg, *pinned; char *uuid; assert_spin_locked(&o2hb_live_lock); - list_for_each_entry(reg, &o2hb_all_regions, hr_all_item) { - if (reg->hr_item_dropped) - continue; + do { + found = 0; + pinned = NULL; - uuid = config_item_name(®->hr_item); + list_for_each_entry(reg, &o2hb_all_regions, hr_all_item) { + if (reg->hr_item_dropped) + continue; - /* local heartbeat */ - if (region_uuid) { - if (strcmp(region_uuid, uuid)) + uuid = config_item_name(®->hr_item); + + /* local heartbeat */ + if (region_uuid) { + if (strcmp(region_uuid, uuid)) + continue; + found = 1; + } + + if (reg->hr_item_pinned || reg->hr_item_dropped) { + if (found) + break; continue; - found = 1; + } + + /* + * Found a region that needs pinning. Take a reference + * so it stays alive while we drop the lock below. + */ + pinned = reg; + config_item_get(®->hr_item); + break; } - if (reg->hr_item_pinned || reg->hr_item_dropped) - goto skip_pin; + if (!pinned) + break; + + uuid = config_item_name(&pinned->hr_item); + + /* + * o2nm_depend_item() -> configfs_depend_item() can sleep (it + * takes the configfs root inode rwsem), so it must not run + * under o2hb_live_lock. Drop the lock across it; @pinned is + * kept alive by the reference taken above. The region list may + * change while unlocked, so we rescan from the top afterwards. + */ + spin_unlock(&o2hb_live_lock); /* Ignore ENOENT only for local hb (userdlm domain) */ - ret = o2nm_depend_item(®->hr_item); + if (from_callback) + ret = o2nm_depend_item_unlocked(&pinned->hr_item); + else + ret = o2nm_depend_item(&pinned->hr_item); + + spin_lock(&o2hb_live_lock); if (!ret) { - mlog(ML_CLUSTER, "Pin region %s\n", uuid); - reg->hr_item_pinned = 1; - } else { - if (ret == -ENOENT && found) - ret = 0; - else { - mlog(ML_ERROR, "Pin region %s fails with %d\n", - uuid, ret); + /* + * o2hb_live_lock was dropped across o2nm_depend_item(). + * o2hb_set_quorum_device() runs in the heartbeat thread + * without o2hb_dependency_mutex, so for global heartbeat + * it may have crossed O2HB_PIN_CUT_OFF and unpinned the + * regions while we slept. If that happened this pin is + * no longer wanted; undo it and stop rather than + * resurrecting it on the rescan below. + */ + if (!region_uuid && + bitmap_weight(o2hb_quorum_region_bitmap, + O2NM_MAX_REGIONS) > O2HB_PIN_CUT_OFF) { + o2nm_undepend_item(&pinned->hr_item); + spin_unlock(&o2hb_live_lock); + config_item_put(&pinned->hr_item); + spin_lock(&o2hb_live_lock); break; } + mlog(ML_CLUSTER, "Pin region %s\n", uuid); + pinned->hr_item_pinned = 1; + } else if (ret == -ENOENT && (found || !region_uuid)) { + /* + * For local hb (found): ignore ENOENT from userdlm + * domains as before. For global hb (!region_uuid): + * the region may have been detached from configfs + * while the lock was dropped — skip it and continue + * pinning the remaining regions. + */ + ret = 0; + } else { + mlog(ML_ERROR, "Pin region %s fails with %d\n", + uuid, ret); } -skip_pin: - if (found) - break; - } + + /* + * config_item_put() may drop the last reference and run + * o2hb_region_release(), which also grabs o2hb_live_lock and + * can sleep, so it must happen with the lock released. + */ + spin_unlock(&o2hb_live_lock); + config_item_put(&pinned->hr_item); + spin_lock(&o2hb_live_lock); + + /* local hb pins a single matching region */ + } while (!ret && !region_uuid); return ret; } @@ -2376,12 +2544,13 @@ static int o2hb_region_inc_user(const char *region_uuid) { int ret = 0; + mutex_lock(&o2hb_dependency_mutex); spin_lock(&o2hb_live_lock); /* local heartbeat */ if (!o2hb_global_heartbeat_active()) { - ret = o2hb_region_pin(region_uuid); - goto unlock; + ret = o2hb_region_pin(region_uuid, false); + goto unlock; } /* @@ -2393,16 +2562,23 @@ static int o2hb_region_inc_user(const char *region_uuid) goto unlock; if (bitmap_weight(o2hb_quorum_region_bitmap, - O2NM_MAX_REGIONS) <= O2HB_PIN_CUT_OFF) - ret = o2hb_region_pin(NULL); + O2NM_MAX_REGIONS) <= O2HB_PIN_CUT_OFF) { + ret = o2hb_region_pin(NULL, false); + if (ret) { + o2hb_region_unpin(NULL); + o2hb_dependent_users--; + } + } unlock: spin_unlock(&o2hb_live_lock); + mutex_unlock(&o2hb_dependency_mutex); return ret; } static void o2hb_region_dec_user(const char *region_uuid) { + mutex_lock(&o2hb_dependency_mutex); spin_lock(&o2hb_live_lock); /* local heartbeat */ @@ -2421,6 +2597,7 @@ static void o2hb_region_dec_user(const char *region_uuid) unlock: spin_unlock(&o2hb_live_lock); + mutex_unlock(&o2hb_dependency_mutex); } int o2hb_register_callback(const char *region_uuid, @@ -2495,7 +2672,7 @@ int o2hb_check_node_heartbeating_no_sem(u8 node_num) unsigned long testing_map[BITS_TO_LONGS(O2NM_MAX_NODES)]; spin_lock(&o2hb_live_lock); - o2hb_fill_node_map_from_callback(testing_map, O2NM_MAX_NODES); + __o2hb_fill_node_map(testing_map, O2NM_MAX_NODES); spin_unlock(&o2hb_live_lock); if (!test_bit(node_num, testing_map)) { mlog(ML_HEARTBEAT, @@ -2512,7 +2689,7 @@ int o2hb_check_node_heartbeating_from_callback(u8 node_num) { unsigned long testing_map[BITS_TO_LONGS(O2NM_MAX_NODES)]; - o2hb_fill_node_map_from_callback(testing_map, O2NM_MAX_NODES); + o2hb_fill_node_map_locked(testing_map, O2NM_MAX_NODES); if (!test_bit(node_num, testing_map)) { mlog(ML_HEARTBEAT, "node (%u) does not have heartbeating enabled.\n", diff --git a/fs/ocfs2/cluster/heartbeat.h b/fs/ocfs2/cluster/heartbeat.h index 8ef8c1b9eeb7..2ca2b657583c 100644 --- a/fs/ocfs2/cluster/heartbeat.h +++ b/fs/ocfs2/cluster/heartbeat.h @@ -58,6 +58,11 @@ int o2hb_register_callback(const char *region_uuid, struct o2hb_callback_func *hc); void o2hb_unregister_callback(const char *region_uuid, struct o2hb_callback_func *hc); +void o2hb_callback_read_lock(void); +void o2hb_callback_read_unlock(void); +void o2hb_synchronize_callbacks(void); +void o2hb_fill_node_map_locked(unsigned long *map, + unsigned int bits); void o2hb_fill_node_map(unsigned long *map, unsigned int bits); void o2hb_exit(void); diff --git a/fs/ocfs2/cluster/nodemanager.c b/fs/ocfs2/cluster/nodemanager.c index 402563154550..e08850a5d736 100644 --- a/fs/ocfs2/cluster/nodemanager.c +++ b/fs/ocfs2/cluster/nodemanager.c @@ -326,6 +326,7 @@ static ssize_t o2nm_node_local_store(struct config_item *item, const char *page, struct o2nm_node *node = to_o2nm_node(item); struct o2nm_cluster *cluster; unsigned long tmp; + bool starting = false; char *p = (char *)page; ssize_t ret; @@ -362,11 +363,13 @@ static ssize_t o2nm_node_local_store(struct config_item *item, const char *page, ret = o2net_start_listening(node); if (ret) goto out; + starting = true; } if (!tmp && cluster->cl_has_local && cluster->cl_local_node == node->nd_num) { o2net_stop_listening(node); + cluster->cl_has_local = 0; cluster->cl_local_node = O2NM_INVALID_NODE_NUM; } @@ -374,6 +377,8 @@ static ssize_t o2nm_node_local_store(struct config_item *item, const char *page, if (node->nd_local) { cluster->cl_has_local = tmp; cluster->cl_local_node = node->nd_num; + if (starting) + o2net_complete_start_listening(node); } ret = count; @@ -777,17 +782,23 @@ int o2nm_depend_item(struct config_item *item) return configfs_depend_item(&o2nm_cluster_group.cs_subsys, item); } +int o2nm_depend_item_unlocked(struct config_item *item) +{ + return configfs_depend_item_unlocked(&o2nm_cluster_group.cs_subsys, + item); +} + void o2nm_undepend_item(struct config_item *item) { configfs_undepend_item(item); } -int o2nm_depend_this_node(void) +int o2nm_depend_node(u8 node_num) { int ret = 0; struct o2nm_node *local_node; - local_node = o2nm_get_node_by_num(o2nm_this_node()); + local_node = o2nm_get_node_by_num(node_num); if (!local_node) { ret = -EINVAL; goto out; @@ -800,17 +811,27 @@ out: return ret; } -void o2nm_undepend_this_node(void) +void o2nm_undepend_node(u8 node_num) { struct o2nm_node *local_node; - local_node = o2nm_get_node_by_num(o2nm_this_node()); + local_node = o2nm_get_node_by_num(node_num); BUG_ON(!local_node); o2nm_undepend_item(&local_node->nd_item); o2nm_node_put(local_node); } +int o2nm_depend_this_node(void) +{ + return o2nm_depend_node(o2nm_this_node()); +} + +void o2nm_undepend_this_node(void) +{ + o2nm_undepend_node(o2nm_this_node()); +} + static void __exit exit_o2nm(void) { diff --git a/fs/ocfs2/cluster/nodemanager.h b/fs/ocfs2/cluster/nodemanager.h index 3490e77a952d..ca3483fb5450 100644 --- a/fs/ocfs2/cluster/nodemanager.h +++ b/fs/ocfs2/cluster/nodemanager.h @@ -64,7 +64,10 @@ void o2nm_node_get(struct o2nm_node *node); void o2nm_node_put(struct o2nm_node *node); int o2nm_depend_item(struct config_item *item); +int o2nm_depend_item_unlocked(struct config_item *item); void o2nm_undepend_item(struct config_item *item); +int o2nm_depend_node(u8 node_num); +void o2nm_undepend_node(u8 node_num); int o2nm_depend_this_node(void); void o2nm_undepend_this_node(void); diff --git a/fs/ocfs2/cluster/tcp.c b/fs/ocfs2/cluster/tcp.c index 09a1f3b77bb8..474fe1414cee 100644 --- a/fs/ocfs2/cluster/tcp.c +++ b/fs/ocfs2/cluster/tcp.c @@ -38,6 +38,8 @@ */ #include <linux/kernel.h> +#include <linux/completion.h> +#include <linux/mutex.h> #include <linux/sched/mm.h> #include <linux/jiffies.h> #include <linux/slab.h> @@ -102,9 +104,16 @@ static struct socket *o2net_listen_sock; * quorum work is queued as sock containers are shutdown.. stop_listening * tears down all the node's sock containers, preventing future shutdowns * and queued quorum work, before canceling delayed quorum work and - * destroying the work queue. + * destroying the work queue. Handler teardown can also race local listener + * shutdown, so keep a waitable destroying pointer until the old ordered + * queue has finished draining. */ static struct workqueue_struct *o2net_wq; +static struct workqueue_struct *o2net_wq_destroying; +static DEFINE_MUTEX(o2net_wq_mutex); +static DECLARE_COMPLETION(o2net_wq_destroyed); +/* Heartbeat callbacks stay registered across local-node off/on. */ +static bool o2net_listening; static struct work_struct o2net_listen_work; static struct o2hb_callback_func o2net_hb_up, o2net_hb_down; @@ -884,6 +893,27 @@ void o2net_unregister_handler_list(struct list_head *list) } EXPORT_SYMBOL_GPL(o2net_unregister_handler_list); +static void o2net_flush_wq(void) +{ + mutex_lock(&o2net_wq_mutex); + if (o2net_wq_destroying) { + mutex_unlock(&o2net_wq_mutex); + wait_for_completion(&o2net_wq_destroyed); + return; + } + + if (o2net_wq) + flush_workqueue(o2net_wq); + mutex_unlock(&o2net_wq_mutex); +} + +void o2net_unregister_and_flush_handler_list(struct list_head *list) +{ + o2net_unregister_handler_list(list); + o2net_flush_wq(); +} +EXPORT_SYMBOL_GPL(o2net_unregister_and_flush_handler_list); + static struct o2net_msg_handler *o2net_handler_get(u32 msg_type, u32 key) { struct o2net_msg_handler *nmh; @@ -1692,6 +1722,19 @@ static void o2net_still_up(struct work_struct *work) /* ------------------------------------------------------------ */ +static void o2net_hb_node_up(struct o2net_node *nn) +{ + /* ensure an immediate connect attempt */ + nn->nn_last_connect_attempt = jiffies - + (msecs_to_jiffies(o2net_reconnect_delay()) + 1); + + spin_lock(&nn->nn_lock); + atomic_set(&nn->nn_timeout, 0); + if (nn->nn_persistent_error) + o2net_set_nn_state(nn, NULL, 0, 0); + spin_unlock(&nn->nn_lock); +} + void o2net_disconnect_node(struct o2nm_node *node) { struct o2net_node *nn = o2net_nn_from_num(node->nd_num); @@ -1702,52 +1745,48 @@ void o2net_disconnect_node(struct o2nm_node *node) o2net_set_nn_state(nn, NULL, 0, -ENOTCONN); spin_unlock(&nn->nn_lock); - if (o2net_wq) { - cancel_delayed_work(&nn->nn_connect_expired); - cancel_delayed_work(&nn->nn_connect_work); - cancel_delayed_work(&nn->nn_still_up); - flush_workqueue(o2net_wq); - } + cancel_delayed_work(&nn->nn_connect_expired); + cancel_delayed_work(&nn->nn_connect_work); + cancel_delayed_work(&nn->nn_still_up); + o2net_flush_wq(); } static void o2net_hb_node_down_cb(struct o2nm_node *node, int node_num, void *data) { + u8 this_node; + o2quo_hb_down(node_num); if (!node) - return; + goto out; + + this_node = o2nm_this_node(); + if (!READ_ONCE(o2net_listening) || this_node == O2NM_MAX_NODES) + goto out; - if (node_num != o2nm_this_node()) + if (node_num != this_node) o2net_disconnect_node(node); +out: BUG_ON(atomic_read(&o2net_connected_peers) < 0); } static void o2net_hb_node_up_cb(struct o2nm_node *node, int node_num, void *data) { - struct o2net_node *nn = o2net_nn_from_num(node_num); + u8 this_node; o2quo_hb_up(node_num); BUG_ON(!node); - /* ensure an immediate connect attempt */ - nn->nn_last_connect_attempt = jiffies - - (msecs_to_jiffies(o2net_reconnect_delay()) + 1); + this_node = o2nm_this_node(); + if (!READ_ONCE(o2net_listening) || this_node == O2NM_MAX_NODES) + return; - if (node_num != o2nm_this_node()) { - /* believe it or not, accept and node heartbeating testing - * can succeed for this node before we got here.. so - * only use set_nn_state to clear the persistent error - * if that hasn't already happened */ - spin_lock(&nn->nn_lock); - atomic_set(&nn->nn_timeout, 0); - if (nn->nn_persistent_error) - o2net_set_nn_state(nn, NULL, 0, 0); - spin_unlock(&nn->nn_lock); - } + if (node_num != this_node) + o2net_hb_node_up(o2net_nn_from_num(node_num)); } void o2net_unregister_hb_callbacks(void) @@ -1756,6 +1795,37 @@ void o2net_unregister_hb_callbacks(void) o2hb_unregister_callback(NULL, &o2net_hb_down); } +/* + * Delay heartbeat-driven network work until the local node is fully published + * through o2nm_this_node(), then replay the nodes that are already live while + * callback delivery stays blocked. + */ +void o2net_complete_start_listening(struct o2nm_node *node) +{ + unsigned long live_nodes[BITS_TO_LONGS(O2NM_MAX_NODES)]; + unsigned long node_num; + u8 local_node; + + local_node = o2nm_this_node(); + if (WARN_ON_ONCE(local_node == O2NM_MAX_NODES)) + return; + if (WARN_ON_ONCE(local_node != node->nd_num)) + return; + if (WARN_ON_ONCE(!o2net_wq)) + return; + + o2hb_callback_read_lock(); + WRITE_ONCE(o2net_listening, true); + o2hb_fill_node_map_locked(live_nodes, O2NM_MAX_NODES); + for_each_set_bit(node_num, live_nodes, O2NM_MAX_NODES) { + if (node_num == local_node) + continue; + + o2net_hb_node_up(o2net_nn_from_num(node_num)); + } + o2hb_callback_read_unlock(); +} + int o2net_register_hb_callbacks(void) { int ret; @@ -2023,6 +2093,36 @@ out: return ret; } +static void o2net_destroy_wq(void) +{ + struct workqueue_struct *wq; + + mutex_lock(&o2net_wq_mutex); + if (o2net_wq_destroying) { + mutex_unlock(&o2net_wq_mutex); + wait_for_completion(&o2net_wq_destroyed); + return; + } + + wq = o2net_wq; + if (!wq) { + mutex_unlock(&o2net_wq_mutex); + return; + } + + reinit_completion(&o2net_wq_destroyed); + o2net_wq_destroying = wq; + mutex_unlock(&o2net_wq_mutex); + + destroy_workqueue(wq); + + mutex_lock(&o2net_wq_mutex); + o2net_wq = NULL; + o2net_wq_destroying = NULL; + complete_all(&o2net_wq_destroyed); + mutex_unlock(&o2net_wq_mutex); +} + /* * called from node manager when we should bring up our network listening * socket. node manager handles all the serialization to only call this @@ -2033,22 +2133,44 @@ out: int o2net_start_listening(struct o2nm_node *node) { int ret = 0; + struct workqueue_struct *wq; + + if (WARN_ON_ONCE(READ_ONCE(o2net_listening))) + return -EBUSY; + + mutex_lock(&o2net_wq_mutex); + if (o2net_wq_destroying) { + mutex_unlock(&o2net_wq_mutex); + return -EBUSY; + } + if (WARN_ON_ONCE(o2net_wq)) { + mutex_unlock(&o2net_wq_mutex); + return -EBUSY; + } + mutex_unlock(&o2net_wq_mutex); - BUG_ON(o2net_wq != NULL); BUG_ON(o2net_listen_sock != NULL); mlog(ML_KTHREAD, "starting o2net thread...\n"); - o2net_wq = alloc_ordered_workqueue("o2net", WQ_MEM_RECLAIM); - if (o2net_wq == NULL) { + wq = alloc_ordered_workqueue("o2net", WQ_MEM_RECLAIM); + if (!wq) { mlog(ML_ERROR, "unable to launch o2net thread\n"); return -ENOMEM; /* ? */ } + mutex_lock(&o2net_wq_mutex); + if (unlikely(o2net_wq_destroying || o2net_wq)) { + mutex_unlock(&o2net_wq_mutex); + destroy_workqueue(wq); + return -EBUSY; + } + o2net_wq = wq; + mutex_unlock(&o2net_wq_mutex); + ret = o2net_open_listening_sock(node->nd_ipv4_address, node->nd_ipv4_port); if (ret) { - destroy_workqueue(o2net_wq); - o2net_wq = NULL; + o2net_destroy_wq(); } else o2quo_conn_up(node->nd_num); @@ -2065,6 +2187,9 @@ void o2net_stop_listening(struct o2nm_node *node) BUG_ON(o2net_wq == NULL); BUG_ON(o2net_listen_sock == NULL); + WRITE_ONCE(o2net_listening, false); + o2hb_synchronize_callbacks(); + /* stop the listening socket from generating work */ write_lock_bh(&sock->sk->sk_callback_lock); sock->sk->sk_data_ready = sock->sk->sk_user_data; @@ -2081,8 +2206,7 @@ void o2net_stop_listening(struct o2nm_node *node) /* finish all work and tear down the work queue */ mlog(ML_KTHREAD, "waiting for o2net thread to exit....\n"); - destroy_workqueue(o2net_wq); - o2net_wq = NULL; + o2net_destroy_wq(); sock_release(o2net_listen_sock); o2net_listen_sock = NULL; diff --git a/fs/ocfs2/cluster/tcp.h b/fs/ocfs2/cluster/tcp.h index a75b551d31c7..a11bcee28947 100644 --- a/fs/ocfs2/cluster/tcp.h +++ b/fs/ocfs2/cluster/tcp.h @@ -89,6 +89,7 @@ int o2net_register_handler(u32 msg_type, u32 key, u32 max_len, o2net_post_msg_handler_func *post_func, struct list_head *unreg_list); void o2net_unregister_handler_list(struct list_head *list); +void o2net_unregister_and_flush_handler_list(struct list_head *list); void o2net_fill_node_map(unsigned long *map, unsigned bytes); @@ -96,6 +97,7 @@ struct o2nm_node; int o2net_register_hb_callbacks(void); void o2net_unregister_hb_callbacks(void); int o2net_start_listening(struct o2nm_node *node); +void o2net_complete_start_listening(struct o2nm_node *node); void o2net_stop_listening(struct o2nm_node *node); void o2net_disconnect_node(struct o2nm_node *node); int o2net_num_connected_peers(void); diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index d7fc3cccf2f4..0075e1624310 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -625,6 +625,28 @@ static int ocfs2_validate_dx_root(struct super_block *sb, le16_to_cpu(el->l_count)); goto bail; } + } else { + struct ocfs2_dx_entry_list *dl_list = &dx_root->dr_entries; + + if (le16_to_cpu(dl_list->de_count) != + ocfs2_dx_entries_per_root(sb)) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid de_count %u (expected %u)\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dl_list->de_count), + ocfs2_dx_entries_per_root(sb)); + goto bail; + } + + if (le16_to_cpu(dl_list->de_num_used) > + le16_to_cpu(dl_list->de_count)) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid de_num_used %u (de_count %u)\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dl_list->de_num_used), + le16_to_cpu(dl_list->de_count)); + goto bail; + } } bail: @@ -664,10 +686,25 @@ static int ocfs2_validate_dx_leaf(struct super_block *sb, return ret; } - if (!OCFS2_IS_VALID_DX_LEAF(dx_leaf)) { - ret = ocfs2_error(sb, "Dir Index Leaf has bad signature %.*s\n", - 7, dx_leaf->dl_signature); - } + if (!OCFS2_IS_VALID_DX_LEAF(dx_leaf)) + return ocfs2_error(sb, "Dir Index Leaf has bad signature %.*s\n", + 7, dx_leaf->dl_signature); + + if (le16_to_cpu(dx_leaf->dl_list.de_count) != + ocfs2_dx_entries_per_leaf(sb)) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has invalid de_count %u (expected %u)\n", + (unsigned long long)le64_to_cpu(dx_leaf->dl_blkno), + le16_to_cpu(dx_leaf->dl_list.de_count), + ocfs2_dx_entries_per_leaf(sb)); + + if (le16_to_cpu(dx_leaf->dl_list.de_num_used) > + le16_to_cpu(dx_leaf->dl_list.de_count)) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has invalid de_num_used %u (de_count %u)\n", + (unsigned long long)le64_to_cpu(dx_leaf->dl_blkno), + le16_to_cpu(dx_leaf->dl_list.de_num_used), + le16_to_cpu(dx_leaf->dl_list.de_count)); return ret; } @@ -1812,7 +1849,12 @@ static int ocfs2_dir_foreach_blk_id(struct inode *inode, * dirent right now. Scan from the start of the block * to make sure. */ if (!inode_eq_iversion(inode, *f_version)) { - for (i = 0; i < i_size_read(inode) && i < offset; ) { + loff_t size = i_size_read(inode); + + for (i = 0; i + OCFS2_DIR_REC_LEN(1) <= size && + i < offset;) { + unsigned int rec_len; + de = (struct ocfs2_dir_entry *) (data->id_data + i); /* It's too expensive to do a full @@ -1821,10 +1863,11 @@ static int ocfs2_dir_foreach_blk_id(struct inode *inode, * least that it is non-zero. A * failure will be detected in the * dirent test below. */ - if (le16_to_cpu(de->rec_len) < - OCFS2_DIR_REC_LEN(1)) + rec_len = le16_to_cpu(de->rec_len); + if (rec_len < OCFS2_DIR_REC_LEN(1) || + i + rec_len > size) break; - i += le16_to_cpu(de->rec_len); + i += rec_len; } ctx->pos = offset = i; *f_version = inode_query_iversion(inode); @@ -1867,6 +1910,7 @@ static int ocfs2_dir_foreach_blk_el(struct inode *inode, struct super_block * sb = inode->i_sb; unsigned int ra_sectors = 16; int stored = 0; + int ret; bh = NULL; @@ -1874,9 +1918,13 @@ static int ocfs2_dir_foreach_blk_el(struct inode *inode, while (ctx->pos < i_size_read(inode)) { blk = ctx->pos >> sb->s_blocksize_bits; - if (ocfs2_read_dir_block(inode, blk, &bh, 0)) { + ret = ocfs2_read_dir_block(inode, blk, &bh, 0); + if (ret) { + if (persist) + return ret; /* Skip the corrupt dirblock and keep trying */ ctx->pos += sb->s_blocksize - offset; + offset = 0; continue; } @@ -1903,7 +1951,10 @@ static int ocfs2_dir_foreach_blk_el(struct inode *inode, * dirent right now. Scan from the start of the block * to make sure. */ if (!inode_eq_iversion(inode, *f_version)) { - for (i = 0; i < sb->s_blocksize && i < offset; ) { + for (i = 0; i + OCFS2_DIR_REC_LEN(1) <= sb->s_blocksize && + i < offset;) { + unsigned int rec_len; + de = (struct ocfs2_dir_entry *) (bh->b_data + i); /* It's too expensive to do a full * dirent test each time round this @@ -1911,13 +1962,14 @@ static int ocfs2_dir_foreach_blk_el(struct inode *inode, * least that it is non-zero. A * failure will be detected in the * dirent test below. */ - if (le16_to_cpu(de->rec_len) < - OCFS2_DIR_REC_LEN(1)) + rec_len = le16_to_cpu(de->rec_len); + if (rec_len < OCFS2_DIR_REC_LEN(1) || + i + rec_len > sb->s_blocksize) break; - i += le16_to_cpu(de->rec_len); + i += rec_len; } offset = i; - ctx->pos = (ctx->pos & ~(sb->s_blocksize - 1)) + ctx->pos = (ctx->pos & ~((loff_t)sb->s_blocksize - 1)) | offset; *f_version = inode_query_iversion(inode); } @@ -1970,8 +2022,7 @@ static int ocfs2_dir_foreach_blk(struct inode *inode, u64 *f_version, int ocfs2_dir_foreach(struct inode *inode, struct dir_context *ctx) { u64 version = inode_query_iversion(inode); - ocfs2_dir_foreach_blk(inode, &version, ctx, true); - return 0; + return ocfs2_dir_foreach_blk(inode, &version, ctx, true); } /* @@ -2168,7 +2219,7 @@ out: /* * routine to check that the specified directory is empty (for rmdir) * - * Returns 1 if dir is empty, zero otherwise. + * Returns 1 if dir is empty, zero if not, and a negative errno on error. * * XXX: This is a performance problem for unindexed directories. */ @@ -2181,8 +2232,10 @@ int ocfs2_empty_dir(struct inode *inode) if (ocfs2_dir_indexed(inode)) { ret = ocfs2_empty_dir_dx(inode, &priv); - if (ret) + if (ret) { mlog_errno(ret); + return ret; + } /* * We still run ocfs2_dir_foreach to get the checks * for "." and "..". @@ -2190,8 +2243,10 @@ int ocfs2_empty_dir(struct inode *inode) } ret = ocfs2_dir_foreach(inode, &priv.ctx); - if (ret) + if (ret) { mlog_errno(ret); + return ret; + } if (!priv.seen_dot || !priv.seen_dot_dot) { mlog(ML_ERROR, "bad directory (dir #%llu) - no `.' or `..'\n", diff --git a/fs/ocfs2/dlm/dlmmaster.c b/fs/ocfs2/dlm/dlmmaster.c index aee3b4c56dcc..612969867ff9 100644 --- a/fs/ocfs2/dlm/dlmmaster.c +++ b/fs/ocfs2/dlm/dlmmaster.c @@ -3099,6 +3099,12 @@ int dlm_migrate_request_handler(struct o2net_msg *msg, u32 len, void *data, name = migrate->name; namelen = migrate->namelen; + if (namelen > DLM_LOCKID_NAME_MAX) { + mlog(ML_ERROR, "%s: invalid name length %u in migrate request\n", + dlm->name, namelen); + ret = -EINVAL; + goto leave; + } hash = dlm_lockid_hash(name, namelen); /* preallocate.. if this fails, abort */ diff --git a/fs/ocfs2/dlm/dlmrecovery.c b/fs/ocfs2/dlm/dlmrecovery.c index 9b97bf73df22..9d4a2695b959 100644 --- a/fs/ocfs2/dlm/dlmrecovery.c +++ b/fs/ocfs2/dlm/dlmrecovery.c @@ -1357,6 +1357,15 @@ int dlm_mig_lockres_handler(struct o2net_msg *msg, u32 len, void *data, if (!dlm_grab(dlm)) return -EINVAL; + if (mres->lockname_len > DLM_LOCKID_NAME_MAX || + mres->num_locks > DLM_MAX_MIGRATABLE_LOCKS || + be16_to_cpu(msg->data_len) < struct_size(mres, ml, mres->num_locks)) { + mlog(ML_ERROR, "%s: invalid lockres migration message from %u\n", + dlm->name, mres->master); + dlm_put(dlm); + return -EINVAL; + } + if (!dlm_joined(dlm)) { mlog(ML_ERROR, "Domain %s not joined! " "lockres %.*s, master %u\n", diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 662dbc845b8b..180107a11046 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1528,6 +1528,22 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } + if ((le32_to_cpu(di->i_flags) & OCFS2_ORPHANED_FL) && + le16_to_cpu(di->i_orphaned_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, "Invalid dinode %llu: orphaned slot %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_orphaned_slot)); + goto bail; + } + + if ((le32_to_cpu(di->i_flags) & OCFS2_DIO_ORPHANED_FL) && + le16_to_cpu(di->i_dio_orphaned_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, "Invalid dinode %llu: DIO orphaned slot %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_dio_orphaned_slot)); + goto bail; + } + /* * Reject dinodes whose i_mode does not name one of the seven * canonical POSIX file types. ocfs2_populate_inode() copies @@ -1608,6 +1624,10 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } + rc = ocfs2_validate_inode_xattr(sb, bh->b_blocknr, di); + if (rc) + goto bail; + if (le16_to_cpu(di->i_dyn_features) & OCFS2_INLINE_DATA_FL) { struct ocfs2_inline_data *data = &di->id2.i_data; @@ -1948,8 +1968,6 @@ int ocfs2_read_inode_block_full(struct inode *inode, struct buffer_head **bh, rc = ocfs2_read_blocks(INODE_CACHE(inode), OCFS2_I(inode)->ip_blkno, 1, &tmp, flags, ocfs2_validate_inode_block); - if (rc < 0) - make_bad_inode(inode); /* If ocfs2_read_blocks() got us a new bh, pass it up. */ if (!rc && !*bh) *bh = tmp; diff --git a/fs/ocfs2/namei.c b/fs/ocfs2/namei.c index b23dd678a7e0..e9c7774ccf91 100644 --- a/fs/ocfs2/namei.c +++ b/fs/ocfs2/namei.c @@ -256,6 +256,7 @@ static int ocfs2_mknod(struct mnt_idmap *idmap, sigset_t oldset; int did_block_signals = 0; struct ocfs2_dentry_lock *dl = NULL; + struct ocfs2_acl_state acl_state = { 0 }; trace_ocfs2_mknod(dir, dentry, dentry->d_name.len, dentry->d_name.name, (unsigned long long)OCFS2_I(dir)->ip_blkno, @@ -330,10 +331,14 @@ static int ocfs2_mknod(struct mnt_idmap *idmap, } } + status = ocfs2_acl_init_prepare(inode, dir, parent_fe_bh, &acl_state); + if (status < 0) + goto leave; + /* calculate meta data/clusters for setting security and acl xattr */ - status = ocfs2_calc_xattr_init(dir, parent_fe_bh, mode, - &si, &want_clusters, - &xattr_credits, &want_meta); + status = ocfs2_calc_xattr_init(dir, mode, &si, &want_clusters, + &xattr_credits, &want_meta, + &acl_state); if (status < 0) { mlog_errno(status); goto leave; @@ -411,8 +416,8 @@ static int ocfs2_mknod(struct mnt_idmap *idmap, inc_nlink(dir); } - status = ocfs2_init_acl(handle, inode, dir, new_fe_bh, parent_fe_bh, - meta_ac, data_ac); + status = ocfs2_init_acl(handle, inode, new_fe_bh, meta_ac, data_ac, + &acl_state); if (status < 0) { mlog_errno(status); @@ -477,6 +482,8 @@ leave: brelse(parent_fe_bh); kfree(si.value); + ocfs2_acl_init_release(&acl_state); + ocfs2_free_dir_lookup_result(&lookup); if (inode_ac) @@ -944,7 +951,10 @@ static int ocfs2_unlink(struct inode *dir, child_locked = 1; if (S_ISDIR(inode->i_mode)) { - if (inode->i_nlink != 2 || !ocfs2_empty_dir(inode)) { + status = ocfs2_empty_dir(inode); + if (status < 0) + goto leave; + if (inode->i_nlink != 2 || !status) { status = -ENOTEMPTY; goto leave; } @@ -1498,8 +1508,10 @@ static int ocfs2_rename(struct mnt_idmap *idmap, if (target_exists) { if (S_ISDIR(new_inode->i_mode)) { - if (new_inode->i_nlink != 2 || - !ocfs2_empty_dir(new_inode)) { + status = ocfs2_empty_dir(new_inode); + if (status < 0) + goto bail; + if (new_inode->i_nlink != 2 || !status) { status = -ENOTEMPTY; goto bail; } @@ -2125,7 +2137,7 @@ static int ocfs2_lookup_lock_orphan_dir(struct ocfs2_super *osb, return ret; } - inode_lock(orphan_dir_inode); + inode_lock_nested(orphan_dir_inode, I_MUTEX_NONDIR2); ret = ocfs2_inode_lock(orphan_dir_inode, &orphan_dir_bh, 1); if (ret < 0) { @@ -2724,7 +2736,7 @@ int ocfs2_del_inode_from_orphan(struct ocfs2_super *osb, goto bail; } - inode_lock(orphan_dir_inode); + inode_lock_nested(orphan_dir_inode, I_MUTEX_NONDIR2); status = ocfs2_inode_lock(orphan_dir_inode, &orphan_dir_bh, 1); if (status < 0) { inode_unlock(orphan_dir_inode); @@ -2837,7 +2849,7 @@ int ocfs2_mv_orphaned_inode_to_new(struct inode *dir, goto leave; } - inode_lock(orphan_dir_inode); + inode_lock_nested(orphan_dir_inode, I_MUTEX_NONDIR2); status = ocfs2_inode_lock(orphan_dir_inode, &orphan_dir_bh, 1); if (status < 0) { diff --git a/fs/ocfs2/refcounttree.c b/fs/ocfs2/refcounttree.c index 7323bde70caa..d9f22b4a2654 100644 --- a/fs/ocfs2/refcounttree.c +++ b/fs/ocfs2/refcounttree.c @@ -116,6 +116,33 @@ static int ocfs2_validate_refcount_block(struct super_block *sb, le32_to_cpu(rb->rf_fs_generation)); goto out; } + + /* + * rf_records (rl_count/rl_used/rl_recs[]) is only meaningful when + * this block is not an interior tree block (OCFS2_REFCOUNT_TREE_FL); + * in that case the same union bytes hold an extent list (rf_list) + * instead, which is validated by ocfs2_validate_extent_block(). + */ + if (!(le32_to_cpu(rb->rf_flags) & OCFS2_REFCOUNT_TREE_FL)) { + if (le16_to_cpu(rb->rf_records.rl_count) != + ocfs2_refcount_recs_per_rb(sb)) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rl_count of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_records.rl_count)); + goto out; + } + + if (le16_to_cpu(rb->rf_records.rl_used) > + le16_to_cpu(rb->rf_records.rl_count)) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rl_used of %u (rl_count %u)\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_records.rl_used), + le16_to_cpu(rb->rf_records.rl_count)); + goto out; + } + } out: return rc; } @@ -3360,10 +3387,9 @@ static int ocfs2_replace_cow(struct ocfs2_cow_context *context) cow_start += num_clusters; } - if (ocfs2_dealloc_has_cluster(&context->dealloc)) { + if (ocfs2_dealloc_has_cluster(&context->dealloc)) ocfs2_schedule_truncate_log_flush(osb, 1); - ocfs2_run_deallocs(osb, &context->dealloc); - } + ocfs2_run_deallocs(osb, &context->dealloc); return ret; } @@ -3846,10 +3872,9 @@ unlock: ocfs2_unlock_refcount_tree(osb, ref_tree, 1); brelse(ref_root_bh); - if (!ret && ocfs2_dealloc_has_cluster(&dealloc)) { + if (!ret && ocfs2_dealloc_has_cluster(&dealloc)) ocfs2_schedule_truncate_log_flush(osb, 1); - ocfs2_run_deallocs(osb, &dealloc); - } + ocfs2_run_deallocs(osb, &dealloc); out: /* * Empty the extent map so that we may get the right extent @@ -4135,10 +4160,9 @@ out_unlock_refcount: ocfs2_unlock_refcount_tree(osb, ref_tree, 1); brelse(ref_root_bh); out: - if (ocfs2_dealloc_has_cluster(&dealloc)) { + if (ocfs2_dealloc_has_cluster(&dealloc)) ocfs2_schedule_truncate_log_flush(osb, 1); - ocfs2_run_deallocs(osb, &dealloc); - } + ocfs2_run_deallocs(osb, &dealloc); return ret; } @@ -4691,10 +4715,9 @@ loff_t ocfs2_reflink_remap_blocks(struct inode *s_inode, } out: - if (ocfs2_dealloc_has_cluster(&dealloc)) { + if (ocfs2_dealloc_has_cluster(&dealloc)) ocfs2_schedule_truncate_log_flush(osb, 1); - ocfs2_run_deallocs(osb, &dealloc); - } + ocfs2_run_deallocs(osb, &dealloc); return ret; } diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index a4a2b87a45fe..20c3aec6b987 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -2759,7 +2759,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, fe->i_clusters = cpu_to_le32(tmp_used - le16_to_cpu(cl->cl_cpg)); spin_lock(&OCFS2_I(alloc_inode)->ip_lock); - OCFS2_I(alloc_inode)->ip_clusters -= le32_to_cpu(fe->i_clusters); + OCFS2_I(alloc_inode)->ip_clusters = le32_to_cpu(fe->i_clusters); fe->i_size = cpu_to_le64(ocfs2_clusters_to_bytes(alloc_inode->i_sb, le32_to_cpu(fe->i_clusters))); spin_unlock(&OCFS2_I(alloc_inode)->ip_lock); diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index fcddd3c13acd..35bcbb0ff607 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -390,6 +390,12 @@ static int ocfs2_init_xattr_bucket(struct ocfs2_xattr_bucket *bucket, return rc; } +static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, + struct ocfs2_xattr_header *xh, + size_t region_size); +static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, + u64 blkno); + /* Read the xattr bucket at xb_blkno */ static int ocfs2_read_xattr_bucket(struct ocfs2_xattr_bucket *bucket, u64 xb_blkno) @@ -408,6 +414,8 @@ static int ocfs2_read_xattr_bucket(struct ocfs2_xattr_bucket *bucket, spin_unlock(&OCFS2_SB(bucket->bu_inode->i_sb)->osb_xattr_lock); if (rc) mlog_errno(rc); + else + rc = ocfs2_validate_xattr_bucket(bucket, xb_blkno); } if (rc) @@ -509,6 +517,22 @@ static int ocfs2_validate_xattr_block(struct super_block *sb, le32_to_cpu(xb->xb_fs_generation)); } + if (!(le16_to_cpu(xb->xb_flags) & OCFS2_XATTR_INDEXED)) { + size_t region_offset = + offsetof(struct ocfs2_xattr_block, xb_attrs.xb_header); + + if (bh->b_size < region_offset) + return ocfs2_error(sb, + "Invalid xattr block %llu: block size %zu is too small\n", + (unsigned long long)bh->b_blocknr, + bh->b_size); + + return ocfs2_validate_xattr_entries_flat(sb, bh->b_blocknr, + &xb->xb_attrs.xb_header, + bh->b_size - + region_offset); + } + return 0; } @@ -611,13 +635,10 @@ int ocfs2_calc_security_init(struct inode *dir, return ret; } -int ocfs2_calc_xattr_init(struct inode *dir, - struct buffer_head *dir_bh, - umode_t mode, +int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, struct ocfs2_security_xattr_info *si, - int *want_clusters, - int *xattr_credits, - int *want_meta) + int *want_clusters, int *xattr_credits, + int *want_meta, struct ocfs2_acl_state *acl_state) { int ret = 0; struct ocfs2_super *osb = OCFS2_SB(dir->i_sb); @@ -628,19 +649,15 @@ int ocfs2_calc_xattr_init(struct inode *dir, si->value_len); if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { - down_read(&OCFS2_I(dir)->ip_xattr_sem); - acl_len = ocfs2_xattr_get_nolock(dir, dir_bh, - OCFS2_XATTR_INDEX_POSIX_ACL_DEFAULT, - "", NULL, 0); - up_read(&OCFS2_I(dir)->ip_xattr_sem); - if (acl_len > 0) { - a_size = ocfs2_xattr_entry_real_size(0, acl_len); - if (S_ISDIR(mode)) - a_size <<= 1; - } else if (acl_len != 0 && acl_len != -ENODATA) { - ret = acl_len; - mlog_errno(ret); - return ret; + if (acl_state->default_acl && S_ISDIR(mode)) { + acl_len = acl_state->default_acl->a_count * + sizeof(struct ocfs2_acl_entry); + a_size += ocfs2_xattr_entry_real_size(0, acl_len); + } + if (acl_state->acl) { + acl_len = acl_state->acl->a_count * + sizeof(struct ocfs2_acl_entry); + a_size += ocfs2_xattr_entry_real_size(0, acl_len); } } @@ -683,14 +700,33 @@ int ocfs2_calc_xattr_init(struct inode *dir, new_clusters); *want_clusters += new_clusters; } - if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL && - acl_len > OCFS2_XATTR_INLINE_SIZE) { - /* for directory, it has DEFAULT and ACCESS two types of acls */ - new_clusters = (S_ISDIR(mode) ? 2 : 1) * - ocfs2_clusters_for_bytes(dir->i_sb, acl_len); - *xattr_credits += ocfs2_clusters_to_blocks(dir->i_sb, - new_clusters); - *want_clusters += new_clusters; + if (osb->s_mount_opt & OCFS2_MOUNT_POSIX_ACL) { + if (acl_state->default_acl && S_ISDIR(mode)) { + acl_len = acl_state->default_acl->a_count * + sizeof(struct ocfs2_acl_entry); + if (acl_len > OCFS2_XATTR_INLINE_SIZE) { + new_clusters = + ocfs2_clusters_for_bytes(dir->i_sb, + acl_len); + *xattr_credits += + ocfs2_clusters_to_blocks(dir->i_sb, + new_clusters); + *want_clusters += new_clusters; + } + } + if (acl_state->acl) { + acl_len = acl_state->acl->a_count * + sizeof(struct ocfs2_acl_entry); + if (acl_len > OCFS2_XATTR_INLINE_SIZE) { + new_clusters = + ocfs2_clusters_for_bytes(dir->i_sb, + acl_len); + *xattr_credits += + ocfs2_clusters_to_blocks(dir->i_sb, + new_clusters); + *want_clusters += new_clusters; + } + } } return ret; @@ -740,12 +776,10 @@ static int ocfs2_xattr_extend_allocation(struct inode *inode, prev_clusters; if (why != RESTART_NONE && clusters_to_add) { - /* - * We can only fail in case the alloc file doesn't give - * up enough clusters. - */ - BUG_ON(why == RESTART_META); - + if (why == RESTART_META) { + status = -ENOSPC; + break; + } credits = ocfs2_calc_extend_credits(inode->i_sb, &vb->vb_xv->xr_list); status = ocfs2_extend_trans(handle, credits); @@ -950,41 +984,206 @@ static int ocfs2_xattr_list_entries(struct inode *inode, return result; } -static int ocfs2_xattr_ibody_lookup_header(struct inode *inode, - struct ocfs2_dinode *di, - struct ocfs2_xattr_header **header) +static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, + struct ocfs2_xattr_header *xh, + size_t region_size) +{ + u16 xattr_count = le16_to_cpu(xh->xh_count); + size_t entries_limit = region_size; + size_t nv_limit = region_size; + size_t max_entries; + int i; + + if (region_size < sizeof(*xh)) + return ocfs2_error(sb, + "Invalid xattr in block %llu: region size %zu is too small\n", + (unsigned long long)blkno, region_size); + + max_entries = (entries_limit - sizeof(*xh)) / + sizeof(struct ocfs2_xattr_entry); + + if (xattr_count > max_entries) + return ocfs2_error(sb, + "Invalid xattr in block %llu: entry count %u exceeds maximum %zu\n", + (unsigned long long)blkno, + xattr_count, max_entries); + + for (i = 0; i < xattr_count; i++) { + struct ocfs2_xattr_entry *xe = &xh->xh_entries[i]; + size_t name_offset = le16_to_cpu(xe->xe_name_offset); + size_t value_offset; + + if (name_offset > nv_limit || + xe->xe_name_len > nv_limit - name_offset) + return ocfs2_error(sb, + "Invalid xattr in block %llu: entry %d name is out of bounds\n", + (unsigned long long)blkno, i); + + value_offset = name_offset + OCFS2_XATTR_SIZE(xe->xe_name_len); + if (value_offset > nv_limit) + return ocfs2_error(sb, + "Invalid xattr in block %llu: entry %d value starts out of bounds\n", + (unsigned long long)blkno, i); + + if (ocfs2_xattr_is_local(xe)) { + if (le64_to_cpu(xe->xe_value_size) > + nv_limit - value_offset) + return ocfs2_error(sb, + "Invalid xattr in block %llu: entry %d value is out of bounds\n", + (unsigned long long)blkno, + i); + } else if (sizeof(struct ocfs2_xattr_value_root) > + nv_limit - value_offset) { + return ocfs2_error(sb, + "Invalid xattr in block %llu: entry %d value root is out of bounds\n", + (unsigned long long)blkno, i); + } + } + + return 0; +} + +static int ocfs2_xattr_ibody_lookup_header_raw(struct super_block *sb, + u64 blkno, + struct ocfs2_dinode *di, + struct ocfs2_xattr_header **header, + u16 *inline_size_ret) { + struct ocfs2_xattr_header *xh; u16 xattr_count; size_t max_entries; u16 inline_size = le16_to_cpu(di->i_xattr_inline_size); - if (inline_size > inode->i_sb->s_blocksize || + if (inline_size > sb->s_blocksize || inline_size < sizeof(struct ocfs2_xattr_header)) { - ocfs2_error(inode->i_sb, - "Invalid xattr inline size %u in inode %llu\n", - inline_size, - (unsigned long long)OCFS2_I(inode)->ip_blkno); + ocfs2_error(sb, + "Invalid inode %llu: xattr inline size %u\n", + (unsigned long long)blkno, inline_size); return -EFSCORRUPTED; } - *header = (struct ocfs2_xattr_header *) - ((void *)di + inode->i_sb->s_blocksize - inline_size); + xh = (struct ocfs2_xattr_header *) + ((void *)di + sb->s_blocksize - inline_size); - xattr_count = le16_to_cpu((*header)->xh_count); + xattr_count = le16_to_cpu(xh->xh_count); max_entries = (inline_size - sizeof(struct ocfs2_xattr_header)) / sizeof(struct ocfs2_xattr_entry); if (xattr_count > max_entries) { - ocfs2_error(inode->i_sb, + ocfs2_error(sb, "xattr entry count %u exceeds maximum %zu in inode %llu\n", xattr_count, max_entries, - (unsigned long long)OCFS2_I(inode)->ip_blkno); + (unsigned long long)blkno); return -EFSCORRUPTED; } + *header = xh; + if (inline_size_ret) + *inline_size_ret = inline_size; + + return 0; +} + +int ocfs2_validate_inode_xattr(struct super_block *sb, u64 blkno, + struct ocfs2_dinode *di) +{ + struct ocfs2_xattr_header *xh; + u16 inline_size; + int ret; + + if (!(le16_to_cpu(di->i_dyn_features) & OCFS2_INLINE_XATTR_FL)) + return 0; + + ret = ocfs2_xattr_ibody_lookup_header_raw(sb, blkno, di, &xh, + &inline_size); + if (ret) + return ret; + + return ocfs2_validate_xattr_entries_flat(sb, blkno, xh, inline_size); +} + +static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, + u64 blkno) +{ + struct super_block *sb = bucket->bu_inode->i_sb; + struct ocfs2_xattr_header *xh = bucket_xh(bucket); + u16 xattr_count = le16_to_cpu(xh->xh_count); + size_t region_size = (size_t)sb->s_blocksize * bucket->bu_blocks; + size_t entries_limit = sb->s_blocksize; + size_t nv_limit = sb->s_blocksize; + size_t max_entries; + int i; + + if (region_size < sizeof(*xh)) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: region size %zu is too small\n", + (unsigned long long)blkno, region_size); + + if (entries_limit < sizeof(*xh)) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entries limit %zu is too small\n", + (unsigned long long)blkno, + entries_limit); + + max_entries = (entries_limit - sizeof(*xh)) / + sizeof(struct ocfs2_xattr_entry); + + if (xattr_count > max_entries) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry count %u exceeds maximum %zu\n", + (unsigned long long)blkno, + xattr_count, max_entries); + + for (i = 0; i < xattr_count; i++) { + struct ocfs2_xattr_entry *xe = &xh->xh_entries[i]; + size_t name_offset = le16_to_cpu(xe->xe_name_offset); + size_t block_off = name_offset >> sb->s_blocksize_bits; + size_t block_offset = name_offset % nv_limit; + size_t value_offset; + + if (name_offset >= region_size || block_off >= bucket->bu_blocks) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry %d name is out of bounds\n", + (unsigned long long)blkno, i); + + if (xe->xe_name_len > nv_limit - block_offset) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry %d name crosses block boundary\n", + (unsigned long long)blkno, i); + + value_offset = block_offset + OCFS2_XATTR_SIZE(xe->xe_name_len); + if (value_offset > nv_limit) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry %d value starts out of bounds\n", + (unsigned long long)blkno, i); + + if (ocfs2_xattr_is_local(xe)) { + if (le64_to_cpu(xe->xe_value_size) > + nv_limit - value_offset) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry %d value is out of bounds\n", + (unsigned long long)blkno, + i); + } else if (sizeof(struct ocfs2_xattr_value_root) > + nv_limit - value_offset) { + return ocfs2_error(sb, + "Invalid xattr bucket %llu: entry %d value root is out of bounds\n", + (unsigned long long)blkno, i); + } + } + return 0; } +static int ocfs2_xattr_ibody_lookup_header(struct inode *inode, + struct ocfs2_dinode *di, + struct ocfs2_xattr_header **header) +{ + return ocfs2_xattr_ibody_lookup_header_raw(inode->i_sb, + OCFS2_I(inode)->ip_blkno, + di, header, NULL); +} + int ocfs2_has_inline_xattr_value_outside(struct inode *inode, struct ocfs2_dinode *di) { @@ -3255,6 +3454,14 @@ meta_guess: credits += OCFS2_SUBALLOC_ALLOC + 1; /* + * Reserve metadata for the new xattr's value extent tree. + * The not_found path above adds credits for this tree but + * omits meta_add, leaving meta_ac NULL for large values. + */ + if (xi->xi_value_len > OCFS2_XATTR_INLINE_SIZE) + meta_add += ocfs2_extend_meta_needed(&def_xv.xv.xr_list); + + /* * This cluster will be used either for new bucket or for * new xattr block. * If the cluster size is the same as the bucket size, one @@ -3483,9 +3690,10 @@ out: } /* - * This function only called duing creating inode - * for init security/acl xattrs of the new inode. - * All transanction credits have been reserved in mknod. + * This helper is only for setting initial ACL or security xattrs on an inode + * that is still unpublished, unhashed, and unattached to a dentry. + * Ordinary xattr updates must use ocfs2_xattr_set(). + * All transaction credits have been reserved in mknod or symlink callers. */ int ocfs2_xattr_set_handle(handle_t *handle, struct inode *inode, @@ -3542,8 +3750,6 @@ int ocfs2_xattr_set_handle(handle_t *handle, xis.inode_bh = xbs.inode_bh = di_bh; di = (struct ocfs2_dinode *)di_bh->b_data; - down_write(&OCFS2_I(inode)->ip_xattr_sem); - ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) goto cleanup; @@ -3556,7 +3762,6 @@ int ocfs2_xattr_set_handle(handle_t *handle, ret = __ocfs2_xattr_set_handle(inode, di, &xi, &xis, &xbs, &ctxt); cleanup: - up_write(&OCFS2_I(inode)->ip_xattr_sem); brelse(xbs.xattr_bh); ocfs2_xattr_bucket_free(xbs.bucket); @@ -7235,10 +7440,9 @@ out_unlock: ref_tree, 1); brelse(ref_root_bh); - if (ocfs2_dealloc_has_cluster(&dealloc)) { + if (ocfs2_dealloc_has_cluster(&dealloc)) ocfs2_schedule_truncate_log_flush(OCFS2_SB(old_inode->i_sb), 1); - ocfs2_run_deallocs(OCFS2_SB(old_inode->i_sb), &dealloc); - } + ocfs2_run_deallocs(OCFS2_SB(old_inode->i_sb), &dealloc); out: return ret; @@ -7257,6 +7461,7 @@ int ocfs2_init_security_and_acl(struct inode *dir, { int ret = 0; struct buffer_head *dir_bh = NULL; + struct ocfs2_acl_state acl_state = { 0 }; ret = ocfs2_init_security_get(inode, dir, qstr, NULL); if (ret) { @@ -7269,10 +7474,17 @@ int ocfs2_init_security_and_acl(struct inode *dir, mlog_errno(ret); goto leave; } - ret = ocfs2_init_acl(NULL, inode, dir, NULL, dir_bh, NULL, NULL); + + ret = ocfs2_acl_init_prepare(inode, dir, dir_bh, &acl_state); + if (ret) + goto unlock; + + ret = ocfs2_init_acl(NULL, inode, NULL, NULL, NULL, &acl_state); if (ret) mlog_errno(ret); +unlock: + ocfs2_acl_init_release(&acl_state); ocfs2_inode_unlock(dir, 0); brelse(dir_bh); leave: diff --git a/fs/ocfs2/xattr.h b/fs/ocfs2/xattr.h index 65e9aa743919..5e18513277f1 100644 --- a/fs/ocfs2/xattr.h +++ b/fs/ocfs2/xattr.h @@ -43,6 +43,8 @@ int ocfs2_xattr_set_handle(handle_t *, struct inode *, struct buffer_head *, struct ocfs2_alloc_context *); int ocfs2_has_inline_xattr_value_outside(struct inode *inode, struct ocfs2_dinode *di); +int ocfs2_validate_inode_xattr(struct super_block *sb, u64 blkno, + struct ocfs2_dinode *di); int ocfs2_xattr_remove(struct inode *, struct buffer_head *); int ocfs2_init_security_get(struct inode *, struct inode *, const struct qstr *, @@ -55,9 +57,12 @@ int ocfs2_init_security_set(handle_t *, struct inode *, int ocfs2_calc_security_init(struct inode *, struct ocfs2_security_xattr_info *, int *, int *, struct ocfs2_alloc_context **); -int ocfs2_calc_xattr_init(struct inode *, struct buffer_head *, - umode_t, struct ocfs2_security_xattr_info *, - int *, int *, int *); + +struct ocfs2_acl_state; +int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, + struct ocfs2_security_xattr_info *si, + int *want_clusters, int *xattr_credits, + int *want_meta, struct ocfs2_acl_state *acl_state); /* * xattrs can live inside an inode, as part of an external xattr block, diff --git a/fs/squashfs/cache.c b/fs/squashfs/cache.c index 67abd4dff222..c1a83bb0e519 100644 --- a/fs/squashfs/cache.c +++ b/fs/squashfs/cache.c @@ -46,18 +46,81 @@ #include "page_actor.h" /* + * Waiters on cache->wait_queue are keyed by the block they want, so a wakeup + * can name who it is for. A NULL key is a capacity wakeup: one entry became + * free, so wake one waiter. A block key is a publication wakeup: that block + * now has an entry, so wake every waiter which can share it. + */ +struct squashfs_cache_wait { + wait_queue_entry_t wait; + u64 block; + bool capacity_wake; +}; + +static int squashfs_cache_wake_function(wait_queue_entry_t *wait, + unsigned int mode, int sync, void *key) +{ + struct squashfs_cache_wait *cache_wait = + container_of(wait, struct squashfs_cache_wait, wait); + u64 *block = key; + + if (block && cache_wait->block != *block) + return 0; + + WRITE_ONCE(cache_wait->capacity_wake, !block); + + /* + * Wake and unlink unconditionally instead of using + * autoremove_wake_function(), which unlinks only when it changed the + * task state. A waiter can be made runnable by something which does + * not go through this queue: wake_up_process() takes TASK_NORMAL, and + * a cgroup v2 thaw calls it on every task in the cgroup, as do + * free_pid() on a pid namespace init and a late rcuwait_wake_up(). + * try_to_wake_up() then fails. Leaving such a waiter queued with a + * reason already recorded would let it act on a freed entry it was not + * given, and the failure would not consume the exclusive budget, so a + * second waiter would be woken for the same entry. + * + * list_del_init_careful() must be the last access to @cache_wait: it + * releases the waiter, whose wait structure lives on its stack, and it + * pairs with list_empty_careful() in finish_wait() to publish the + * store above. __wake_up_common() samples ->flags and the next entry + * before calling here, so it does not touch @wait afterwards either. + */ + default_wake_function(wait, mode, sync, key); + list_del_init_careful(&wait->entry); + + return 1; +} + +static void squashfs_cache_wake_block(struct squashfs_cache *cache, u64 block) +{ + /* nr_exclusive == 0: wake every waiter which matches the key. */ + __wake_up(&cache->wait_queue, TASK_NORMAL, 0, &block); +} + +/* * Look-up block in cache, and increment usage count. If not in cache, read * and decompress it from disk. + * + * A caller which finds no free entry sleeps on cache->wait_queue as an + * exclusive waiter, so squashfs_cache_put() releasing one entry wakes exactly + * one task. Because a wakee may find its block published in the meantime and + * share that entry rather than claim the free one, a wakee which shares hands + * its wakeup on to the next waiter. */ struct squashfs_cache_entry *squashfs_cache_get(struct super_block *sb, struct squashfs_cache *cache, u64 block, int length) { int i, n; struct squashfs_cache_entry *entry; + bool capacity_wake = false; spin_lock(&cache->lock); while (1) { + bool pending, wake_next, wake_block; + for (i = cache->curr_blk, n = 0; n < cache->entries; n++) { if (cache->entry[i].block == block) { cache->curr_blk = i; @@ -72,9 +135,25 @@ struct squashfs_cache_entry *squashfs_cache_get(struct super_block *sb, * go to sleep waiting for one to become available. */ if (cache->unused == 0) { + struct squashfs_cache_wait wait = { + .block = block, + .capacity_wake = false, + }; + + init_wait_func(&wait.wait, + squashfs_cache_wake_function); cache->num_waiters++; + /* + * Enqueue while still holding cache->lock, so + * that a concurrent lookup either sees us + * queued or we see the block it publishes. + */ + prepare_to_wait_exclusive(&cache->wait_queue, + &wait.wait, TASK_UNINTERRUPTIBLE); spin_unlock(&cache->lock); - wait_event(cache->wait_queue, cache->unused); + schedule(); + finish_wait(&cache->wait_queue, &wait.wait); + capacity_wake = READ_ONCE(wait.capacity_wake); spin_lock(&cache->lock); cache->num_waiters--; continue; @@ -105,8 +184,18 @@ struct squashfs_cache_entry *squashfs_cache_get(struct super_block *sb, entry->pending = 1; entry->num_waiters = 0; entry->error = 0; + wake_block = cache->num_waiters > 0; spin_unlock(&cache->lock); + /* + * The entry is now findable, so release everybody + * queued for this block to share it rather than each + * waiting for an entry of their own. They will block + * on entry->wait_queue below until the read completes. + */ + if (wake_block) + squashfs_cache_wake_block(cache, block); + entry->length = squashfs_read_data(sb, block, length, &entry->next_index, entry->actor); @@ -138,20 +227,33 @@ struct squashfs_cache_entry *squashfs_cache_get(struct super_block *sb, * for reuse. */ entry = &cache->entry[i]; - if (entry->refcount == 0) + if (entry->refcount == 0) { cache->unused--; + /* This claims the capacity we were woken for. */ + capacity_wake = false; + } entry->refcount++; /* * If the entry is currently being filled in by another process * go to sleep waiting for it to become available. */ - if (entry->pending) { + pending = entry->pending; + if (pending) entry->num_waiters++; - spin_unlock(&cache->lock); + + /* + * We were woken because an entry became free, but shared a + * block instead of claiming it. Hand the wakeup on, otherwise + * the free entry sits unclaimed while others sleep. + */ + wake_next = capacity_wake && cache->unused && cache->num_waiters; + spin_unlock(&cache->lock); + + if (wake_next) + wake_up(&cache->wait_queue); + if (pending) wait_event(entry->wait_queue, !entry->pending); - } else - spin_unlock(&cache->lock); goto out; } @@ -299,7 +401,7 @@ int squashfs_copy_data(void *buffer, struct squashfs_cache_entry *entry, { int remaining = length; - if (length == 0) + if (length == 0 || offset < 0) return 0; else if (buffer == NULL) return min(length, entry->length - offset); diff --git a/fs/squashfs/squashfs_fs_sb.h b/fs/squashfs/squashfs_fs_sb.h index c01998eec146..b87221ea9bdd 100644 --- a/fs/squashfs/squashfs_fs_sb.h +++ b/fs/squashfs/squashfs_fs_sb.h @@ -12,6 +12,15 @@ #include "squashfs_fs.h" +/* + * Waiters for a cache entry sleep on wait_queue as exclusive waiters, so + * freeing one entry wakes one task. See squashfs_cache_get(). + * + * num_waiters is only a hint used to skip pointless wakeups: it is + * incremented before a task queues itself and decremented after it is woken, + * so it can transiently exceed the number of queued tasks. It never + * undercounts them, which is what the wakeup paths rely on. + */ struct squashfs_cache { char *name; int entries; diff --git a/include/linux/list.h b/include/linux/list.h index 19212bfc3f6d..77fb62f79928 100644 --- a/include/linux/list.h +++ b/include/linux/list.h @@ -158,10 +158,13 @@ static inline bool __list_del_entry_valid(struct list_head *entry) * * This is only for internal list manipulation where we know * the prev/next entries already! + * + * Must be inlined to ensure it can be safely called + * with initdata arguments. */ -static inline void __list_add(struct list_head *new, - struct list_head *prev, - struct list_head *next) +static __always_inline void __list_add(struct list_head *new, + struct list_head *prev, + struct list_head *next) { if (!__list_add_valid(new, prev, next)) return; @@ -179,8 +182,12 @@ static inline void __list_add(struct list_head *new, * * Insert a new entry after the specified head. * This is good for implementing stacks. + * + * Must be inlined to ensure it can be safely called + * with initdata arguments. */ -static inline void list_add(struct list_head *new, struct list_head *head) +static __always_inline void list_add(struct list_head *new, + struct list_head *head) { __list_add(new, head, head->next); } diff --git a/include/linux/llist.h b/include/linux/llist.h index 8846b7709669..413249764f7b 100644 --- a/include/linux/llist.h +++ b/include/linux/llist.h @@ -66,7 +66,7 @@ struct llist_node { /** * init_llist_head - initialize lock-less list head - * @head: the head for your lock-less list + * @list: the head for your lock-less list */ static inline void init_llist_head(struct llist_head *list) { diff --git a/include/linux/nmi.h b/include/linux/nmi.h index a394656bdb6d..af69712df5f4 100644 --- a/include/linux/nmi.h +++ b/include/linux/nmi.h @@ -77,9 +77,9 @@ static inline void reset_hung_task_detector(void) { } * detectors are 'suspended' while 'watchdog_thresh' is equal zero. */ #define WATCHDOG_HARDLOCKUP_ENABLED_BIT 0 -#define WATCHDOG_SOFTOCKUP_ENABLED_BIT 1 +#define WATCHDOG_SOFTLOCKUP_ENABLED_BIT 1 #define WATCHDOG_HARDLOCKUP_ENABLED (1 << WATCHDOG_HARDLOCKUP_ENABLED_BIT) -#define WATCHDOG_SOFTOCKUP_ENABLED (1 << WATCHDOG_SOFTOCKUP_ENABLED_BIT) +#define WATCHDOG_SOFTLOCKUP_ENABLED (1 << WATCHDOG_SOFTLOCKUP_ENABLED_BIT) #if defined(CONFIG_HARDLOCKUP_DETECTOR) extern void hardlockup_detector_disable(void); diff --git a/include/linux/pps_kernel.h b/include/linux/pps_kernel.h index 9f088c9023b1..00b840970d56 100644 --- a/include/linux/pps_kernel.h +++ b/include/linux/pps_kernel.h @@ -60,6 +60,7 @@ struct pps_device { struct device dev; struct fasync_struct *async_queue; /* fasync method */ spinlock_t lock; + bool kc_removed; }; /* diff --git a/include/linux/rbtree.h b/include/linux/rbtree.h index 48acdc3889dd..b7078b0e1eb2 100644 --- a/include/linux/rbtree.h +++ b/include/linux/rbtree.h @@ -500,7 +500,7 @@ rb_find_first(const void *key, const struct rb_root *tree, /** * rb_next_match() - find the next @key in @tree * @key: key to match - * @tree: tree to search + * @node: tree to search * @cmp: operator defining node order * * Returns the next node matching @key, or NULL. diff --git a/include/linux/rbtree_latch.h b/include/linux/rbtree_latch.h index 2f630eb8307e..b55a3a01bc8a 100644 --- a/include/linux/rbtree_latch.h +++ b/include/linux/rbtree_latch.h @@ -47,7 +47,7 @@ struct latch_tree_root { }; /** - * latch_tree_ops - operators to define the tree order + * struct latch_tree_ops - operators to define the tree order * @less: used for insertion; provides the (partial) order between two elements. * @comp: used for lookups; provides the order between the search key and an element. * diff --git a/include/linux/sched.h b/include/linux/sched.h index 3f100d69b053..c7cdbb3e4aa6 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1554,6 +1554,14 @@ struct task_struct { /* Collect coverage from softirq context: */ unsigned int kcov_softirq; + + /* Temporary storage for preempting remote coverage collection: */ + unsigned int kcov_saved_mode; + unsigned int kcov_saved_size; + void *kcov_saved_area; + struct kcov *kcov_saved_kcov; + int kcov_saved_sequence; + #endif #ifdef CONFIG_MEMCG_V1 diff --git a/include/linux/stacktrace.h b/include/linux/stacktrace.h index 97455880ac41..525cf60673fe 100644 --- a/include/linux/stacktrace.h +++ b/include/linux/stacktrace.h @@ -11,7 +11,7 @@ struct pt_regs; #ifdef CONFIG_ARCH_STACKWALK /** - * stack_trace_consume_fn - Callback for arch_stack_walk() + * typedef stack_trace_consume_fn - Callback for arch_stack_walk() * @cookie: Caller supplied pointer handed back by arch_stack_walk() * @addr: The stack entry address to consume * @@ -48,9 +48,9 @@ void arch_stack_walk(stack_trace_consume_fn consume_entry, void *cookie, * @consume_entry * @task: Pointer to a task struct, can be NULL * - * This function returns an error if it detects any unreliable + * Returns: a negative error code if it detects any unreliable * features of the stack. Otherwise it guarantees that the stack - * trace is reliable. + * trace is reliable and returns %0. * * If the task is not 'current', the caller *must* ensure the task is * inactive and its stack is pinned. diff --git a/include/linux/taskstats_kern.h b/include/linux/taskstats_kern.h index dbb4d124c7d7..995cd19b56c1 100644 --- a/include/linux/taskstats_kern.h +++ b/include/linux/taskstats_kern.h @@ -14,7 +14,6 @@ #ifdef CONFIG_TASKSTATS extern struct kmem_cache *taskstats_cache; -extern struct mutex taskstats_exit_mutex; static inline void taskstats_tgid_free(struct signal_struct *sig) { diff --git a/ipc/ipc_sysctl.c b/ipc/ipc_sysctl.c index 9b087ebeb643..d038d944257f 100644 --- a/ipc/ipc_sysctl.c +++ b/ipc/ipc_sysctl.c @@ -28,7 +28,7 @@ static int proc_ipc_dointvec_minmax_orphans(const struct ctl_table *table, int w if (err < 0) return err; - if (ns->shm_rmid_forced) + if (write && ns->shm_rmid_forced) shm_destroy_orphaned(ns); return err; } diff --git a/kernel/fork.c b/kernel/fork.c index e645675dd727..416758c8a3d4 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -143,7 +143,7 @@ unsigned long total_forks; /* Handle normal Linux uptimes. */ int nr_threads; /* The idle threads do not count.. */ -static int max_threads; /* tunable limit on nr_threads */ +static int max_threads __read_mostly; /* tunable limit on nr_threads */ #define NAMED_ARRAY_INDEX(x) [x] = __stringify(x) diff --git a/kernel/kcov.c b/kernel/kcov.c index b19b473c366a..35420f0ac524 100644 --- a/kernel/kcov.c +++ b/kernel/kcov.c @@ -86,17 +86,12 @@ struct kcov_remote { static DEFINE_SPINLOCK(kcov_remote_lock); static DEFINE_HASHTABLE(kcov_remote_map, 4); -static struct list_head kcov_remote_areas = LIST_HEAD_INIT(kcov_remote_areas); +static struct list_head kcov_remote_areas[2] = { + LIST_HEAD_INIT(kcov_remote_areas[0]), LIST_HEAD_INIT(kcov_remote_areas[1]) +}; struct kcov_percpu_data { - void *irq_area; local_lock_t lock; - - unsigned int saved_mode; - unsigned int saved_size; - void *saved_area; - struct kcov *saved_kcov; - int saved_sequence; }; static DEFINE_PER_CPU(struct kcov_percpu_data, kcov_percpu_data) = { @@ -132,12 +127,13 @@ static struct kcov_remote *kcov_remote_add(struct kcov *kcov, u64 handle) } /* Must be called with kcov_remote_lock locked. */ -static struct kcov_remote_area *kcov_remote_area_get(unsigned int size) +static struct kcov_remote_area *kcov_remote_area_get(unsigned int size, bool irq) { struct kcov_remote_area *area; struct list_head *pos; + struct list_head *list = &kcov_remote_areas[irq]; - list_for_each(pos, &kcov_remote_areas) { + list_for_each(pos, list) { area = list_entry(pos, struct kcov_remote_area, list); if (area->size == size) { list_del(&area->list); @@ -149,11 +145,11 @@ static struct kcov_remote_area *kcov_remote_area_get(unsigned int size) /* Must be called with kcov_remote_lock locked. */ static void kcov_remote_area_put(struct kcov_remote_area *area, - unsigned int size) + unsigned int size, bool irq) { INIT_LIST_HEAD(&area->list); area->size = size; - list_add(&area->list, &kcov_remote_areas); + list_add(&area->list, &kcov_remote_areas[irq]); /* * KMSAN doesn't instrument this file, so it may not know area->list * is initialized. Unpoison it explicitly to avoid reports in @@ -390,6 +386,12 @@ void kcov_task_init(struct task_struct *t) kcov_task_reset(t); t->kcov_remote = NULL; t->kcov_handle = current->kcov_handle; + t->kcov_softirq = 0; + t->kcov_saved_mode = 0; + t->kcov_saved_size = 0; + t->kcov_saved_area = NULL; + t->kcov_saved_kcov = NULL; + t->kcov_saved_sequence = 0; } static void kcov_reset(struct kcov *kcov) @@ -836,17 +838,16 @@ static inline bool kcov_mode_enabled(unsigned int mode) static void kcov_remote_softirq_start(struct task_struct *t) __must_hold(&kcov_percpu_data.lock) { - struct kcov_percpu_data *data = this_cpu_ptr(&kcov_percpu_data); unsigned int mode; mode = READ_ONCE(t->kcov_mode); barrier(); if (kcov_mode_enabled(mode)) { - data->saved_mode = mode; - data->saved_size = t->kcov_size; - data->saved_area = t->kcov_area; - data->saved_sequence = t->kcov_sequence; - data->saved_kcov = t->kcov; + t->kcov_saved_mode = mode; + t->kcov_saved_size = t->kcov_size; + t->kcov_saved_area = t->kcov_area; + t->kcov_saved_sequence = t->kcov_sequence; + t->kcov_saved_kcov = t->kcov; kcov_stop(t); } } @@ -854,17 +855,15 @@ static void kcov_remote_softirq_start(struct task_struct *t) static void kcov_remote_softirq_stop(struct task_struct *t) __must_hold(&kcov_percpu_data.lock) { - struct kcov_percpu_data *data = this_cpu_ptr(&kcov_percpu_data); - - if (data->saved_kcov) { - kcov_start(t, data->saved_kcov, data->saved_size, - data->saved_area, data->saved_mode, - data->saved_sequence); - data->saved_mode = 0; - data->saved_size = 0; - data->saved_area = NULL; - data->saved_sequence = 0; - data->saved_kcov = NULL; + if (t->kcov_saved_kcov) { + kcov_start(t, t->kcov_saved_kcov, t->kcov_saved_size, + t->kcov_saved_area, t->kcov_saved_mode, + t->kcov_saved_sequence); + t->kcov_saved_mode = 0; + t->kcov_saved_size = 0; + t->kcov_saved_area = NULL; + t->kcov_saved_sequence = 0; + t->kcov_saved_kcov = NULL; } } @@ -927,17 +926,17 @@ void kcov_remote_start(u64 handle) sequence = kcov->sequence; if (in_task()) { size = kcov->remote_size; - area = kcov_remote_area_get(size); + area = kcov_remote_area_get(size, false); } else { size = CONFIG_KCOV_IRQ_AREA_SIZE; - area = this_cpu_ptr(&kcov_percpu_data)->irq_area; + area = kcov_remote_area_get(size, true); } spin_unlock(&kcov_remote_lock); - /* Can only happen when in_task(). */ + /* Allocate new buffer if we can sleep. */ if (!area) { local_unlock_irqrestore(&kcov_percpu_data.lock, flags); - area = vmalloc(size * sizeof(unsigned long)); + area = in_task() ? vmalloc(size * sizeof(unsigned long)) : NULL; if (!area) { kcov_put(kcov); return; @@ -1079,11 +1078,9 @@ void kcov_remote_stop(void) kcov_move_area(kcov->mode, kcov->area, kcov->size, area); spin_unlock(&kcov->lock); - if (in_task()) { - spin_lock(&kcov_remote_lock); - kcov_remote_area_put(area, size); - spin_unlock(&kcov_remote_lock); - } + spin_lock(&kcov_remote_lock); + kcov_remote_area_put(area, size, !in_task()); + spin_unlock(&kcov_remote_lock); local_unlock_irqrestore(&kcov_percpu_data.lock, flags); @@ -1129,14 +1126,21 @@ static void __init selftest(void) static int __init kcov_init(void) { - int cpu; + int cpu = num_possible_cpus(); + +#ifdef CONFIG_PREEMPT_RT + /* Allocate some extra buffers in order to prepare for softirq preemption. */ + cpu = cpu >= 4 ? cpu * 2 : cpu + 4; +#endif + while (cpu--) { + void *area = vmalloc(CONFIG_KCOV_IRQ_AREA_SIZE * sizeof(unsigned long)); + unsigned long flags; - for_each_possible_cpu(cpu) { - void *area = vmalloc_node(CONFIG_KCOV_IRQ_AREA_SIZE * - sizeof(unsigned long), cpu_to_node(cpu)); if (!area) return -ENOMEM; - per_cpu_ptr(&kcov_percpu_data, cpu)->irq_area = area; + spin_lock_irqsave(&kcov_remote_lock, flags); + kcov_remote_area_put(area, CONFIG_KCOV_IRQ_AREA_SIZE, true); + spin_unlock_irqrestore(&kcov_remote_lock, flags); } /* diff --git a/kernel/params.c b/kernel/params.c index a668863a4bb6..5c8a05921e28 100644 --- a/kernel/params.c +++ b/kernel/params.c @@ -136,8 +136,8 @@ static int parse_one(char *param, if (!val && !(params[i].ops->flags & KERNEL_PARAM_OPS_FL_NOARG)) return -EINVAL; - pr_debug("handling %s with %p\n", param, - params[i].ops->set); + pr_debug("handling %s with value '%s'\n", param, + val ? val : "no-arg"); kernel_param_lock(params[i].mod); if (param_check_unsafe(¶ms[i])) err = params[i].ops->set(val, ¶ms[i]); diff --git a/kernel/resource.c b/kernel/resource.c index 3d17e3196a3e..e60539a55541 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1859,7 +1859,7 @@ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) if (p->flags & IORESOURCE_BUSY) continue; - pr_warn("resource sanity check: requesting [mem %pa-%pa], which spans more than %s %pR\n", + pr_debug("resource sanity check: requesting [mem %pa-%pa], which spans more than %s %pR\n", &addr, &end, p->name, p); err = -1; break; diff --git a/kernel/signal.c b/kernel/signal.c index bbc0fd4cc4d7..a5e15bf09d31 100644 --- a/kernel/signal.c +++ b/kernel/signal.c @@ -130,28 +130,10 @@ static bool sig_ignored(struct task_struct *t, int sig, bool force) */ static inline bool has_pending_signals(sigset_t *signal, sigset_t *blocked) { - unsigned long ready; - long i; - - switch (_NSIG_WORDS) { - default: - for (i = _NSIG_WORDS, ready = 0; --i >= 0 ;) - ready |= signal->sig[i] &~ blocked->sig[i]; - break; - - case 4: ready = signal->sig[3] &~ blocked->sig[3]; - ready |= signal->sig[2] &~ blocked->sig[2]; - ready |= signal->sig[1] &~ blocked->sig[1]; - ready |= signal->sig[0] &~ blocked->sig[0]; - break; - - case 2: ready = signal->sig[1] &~ blocked->sig[1]; - ready |= signal->sig[0] &~ blocked->sig[0]; - break; - - case 1: ready = signal->sig[0] &~ blocked->sig[0]; - } - return ready != 0; + unsigned long ready = 0; + for (long i = 0; i < _NSIG_WORDS; i++) + ready |= signal->sig[i] & ~blocked->sig[i]; + return ready != 0; } #define PENDING(p,b) has_pending_signals(&(p)->signal, (b)) @@ -1181,6 +1163,7 @@ static inline bool has_si_pid_and_uid(struct kernel_siginfo *info) int send_signal_locked(int sig, struct kernel_siginfo *info, struct task_struct *t, enum pid_type type) { + struct kernel_siginfo __maybe_unused rewritten; /* Should SIGKILL or SIGSTOP be received by a pid namespace init? */ bool force = false; @@ -1192,24 +1175,34 @@ int send_signal_locked(int sig, struct kernel_siginfo *info, force = true; } else if (has_si_pid_and_uid(info)) { /* SIGKILL and SIGSTOP is special or has ids */ +#ifdef CONFIG_USER_NS struct user_namespace *t_user_ns; + kuid_t uid; rcu_read_lock(); t_user_ns = task_cred_xxx(t, user_ns); if (current_user_ns() != t_user_ns) { - kuid_t uid = make_kuid(current_user_ns(), info->si_uid); - info->si_uid = from_kuid_munged(t_user_ns, uid); + rewritten = *info; + info = &rewritten; + uid = make_kuid(current_user_ns(), info->si_uid); + rewritten.si_uid = from_kuid_munged(t_user_ns, uid); } rcu_read_unlock(); - +#endif /* A kernel generated signal? */ force = (info->si_code == SI_KERNEL); +#ifdef CONFIG_PID_NS /* From an ancestor pid namespace? */ if (!task_pid_nr_ns(current, task_active_pid_ns(t))) { - info->si_pid = 0; + if (info != &rewritten) { + rewritten = *info; + info = &rewritten; + } + rewritten.si_pid = 0; force = true; } +#endif } return __send_signal_locked(sig, info, t, type, force); } @@ -3951,6 +3944,15 @@ static void prepare_kill_siginfo(int sig, struct kernel_siginfo *info, info->si_uid = from_kuid_munged(current_user_ns(), current_uid()); } +/* + * Not even root can pretend to send SI_FROMKERNEL() signals. + * Nor can they impersonate kill()/tgkill(), which have si_pid/uid + */ +static bool si_code_reserved_to_kernel(int si_code) +{ + return si_code >= 0 || si_code == SI_TKILL; +} + /** * sys_kill - send a signal to a process * @pid: the PID of the process @@ -3958,11 +3960,7 @@ static void prepare_kill_siginfo(int sig, struct kernel_siginfo *info, */ SYSCALL_DEFINE2(kill, pid_t, pid, int, sig) { - struct kernel_siginfo info; - - prepare_kill_siginfo(sig, &info, PIDTYPE_TGID); - - return kill_something_info(sig, &info, pid); + return kill_something_info(sig, SEND_SIG_NOINFO, pid); } /* @@ -4046,7 +4044,7 @@ static int do_pidfd_send_signal(struct pid *pid, int sig, enum pid_type type, /* Only allow sending arbitrary signals to yourself. */ if ((task_pid(current) != pid || type > PIDTYPE_TGID) && - (kinfo.si_code >= 0 || kinfo.si_code == SI_TKILL)) + si_code_reserved_to_kernel(kinfo.si_code)) return -EPERM; } else { prepare_kill_siginfo(sig, &kinfo, type); @@ -4201,11 +4199,8 @@ SYSCALL_DEFINE2(tkill, pid_t, pid, int, sig) static int do_rt_sigqueueinfo(pid_t pid, int sig, kernel_siginfo_t *info) { - /* Not even root can pretend to send signals from the kernel. - * Nor can they impersonate a kill()/tgkill(), which adds source info. - */ - if ((info->si_code >= 0 || info->si_code == SI_TKILL) && - (task_pid_vnr(current) != pid)) + if (si_code_reserved_to_kernel(info->si_code) && + task_pid_vnr(current) != pid) return -EPERM; /* POSIX.1b doesn't mention process groups. */ @@ -4248,11 +4243,8 @@ static int do_rt_tgsigqueueinfo(pid_t tgid, pid_t pid, int sig, kernel_siginfo_t if (pid <= 0 || tgid <= 0) return -EINVAL; - /* Not even root can pretend to send signals from the kernel. - * Nor can they impersonate a kill()/tgkill(), which adds source info. - */ - if ((info->si_code >= 0 || info->si_code == SI_TKILL) && - (task_pid_vnr(current) != pid)) + if (si_code_reserved_to_kernel(info->si_code) && + task_pid_vnr(current) != pid) return -EPERM; return do_send_specific(tgid, pid, sig, info); diff --git a/kernel/sys.c b/kernel/sys.c index df69bd71de03..35b538ba843c 100644 --- a/kernel/sys.c +++ b/kernel/sys.c @@ -2189,7 +2189,7 @@ static int prctl_set_auxv(struct mm_struct *mm, unsigned long addr, BUILD_BUG_ON(sizeof(user_auxv) != sizeof(mm->saved_auxv)); task_lock(current); - memcpy(mm->saved_auxv, user_auxv, len); + memcpy(mm->saved_auxv, user_auxv, sizeof(user_auxv)); task_unlock(current); return 0; diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 2cd0172d0516..f31df72f0e9d 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -361,17 +361,14 @@ static int parse(struct nlattr *na, struct cpumask *mask) int len; int ret; - if (na == NULL) - return 1; len = nla_len(na); if (len > TASKSTATS_CPUMASK_MAXLEN) return -E2BIG; if (len < 1) return -EINVAL; - data = kmalloc(len, GFP_KERNEL); + data = nla_strdup(na, GFP_KERNEL); if (!data) return -ENOMEM; - nla_strscpy(data, na, len); ret = cpulist_parse(data, mask); kfree(data); return ret; @@ -423,7 +420,7 @@ static int cgroupstats_user_cmd(struct sk_buff *skb, struct genl_info *info) fd = nla_get_u32(info->attrs[CGROUPSTATS_CMD_ATTR_FD]); CLASS(fd, f)(fd); if (fd_empty(f)) - return 0; + return -EBADF; size = nla_total_size(sizeof(struct cgroupstats)); @@ -451,36 +448,18 @@ static int cgroupstats_user_cmd(struct sk_buff *skb, struct genl_info *info) return send_reply(rep_skb, info); } -static int cmd_attr_register_cpumask(struct genl_info *info) -{ - cpumask_var_t mask; - int rc; - - if (!alloc_cpumask_var(&mask, GFP_KERNEL)) - return -ENOMEM; - rc = parse(info->attrs[TASKSTATS_CMD_ATTR_REGISTER_CPUMASK], mask); - if (rc < 0) - goto out; - rc = add_del_listener(info->snd_portid, mask, REGISTER); -out: - free_cpumask_var(mask); - return rc; -} - -static int cmd_attr_deregister_cpumask(struct genl_info *info) +static int cmd_attr_cpumask(struct genl_info *info, int attr, + enum actions action) { - cpumask_var_t mask; + cpumask_var_t mask __free(free_cpumask_var) = CPUMASK_VAR_NULL; int rc; if (!alloc_cpumask_var(&mask, GFP_KERNEL)) return -ENOMEM; - rc = parse(info->attrs[TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK], mask); + rc = parse(info->attrs[attr], mask); if (rc < 0) - goto out; - rc = add_del_listener(info->snd_portid, mask, DEREGISTER); -out: - free_cpumask_var(mask); - return rc; + return rc; + return add_del_listener(info->snd_portid, mask, action); } static size_t taskstats_packet_size(void) @@ -555,9 +534,13 @@ err: static int taskstats_user_cmd(struct sk_buff *skb, struct genl_info *info) { if (info->attrs[TASKSTATS_CMD_ATTR_REGISTER_CPUMASK]) - return cmd_attr_register_cpumask(info); + return cmd_attr_cpumask(info, + TASKSTATS_CMD_ATTR_REGISTER_CPUMASK, + REGISTER); else if (info->attrs[TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK]) - return cmd_attr_deregister_cpumask(info); + return cmd_attr_cpumask(info, + TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK, + DEREGISTER); else if (info->attrs[TASKSTATS_CMD_ATTR_PID]) return cmd_attr_pid(info); else if (info->attrs[TASKSTATS_CMD_ATTR_TGID]) diff --git a/kernel/watchdog.c b/kernel/watchdog.c index 87dd5e0f6968..e5134ad7b663 100644 --- a/kernel/watchdog.c +++ b/kernel/watchdog.c @@ -359,14 +359,14 @@ static void lockup_detector_update_enable(void) if (watchdog_hardlockup_available && watchdog_hardlockup_user_enabled) watchdog_enabled |= WATCHDOG_HARDLOCKUP_ENABLED; if (watchdog_softlockup_user_enabled) - watchdog_enabled |= WATCHDOG_SOFTOCKUP_ENABLED; + watchdog_enabled |= WATCHDOG_SOFTLOCKUP_ENABLED; } #ifdef CONFIG_SOFTLOCKUP_DETECTOR /* - * Delay the soflockup report when running a known slow code. - * It does _not_ affect the timestamp of the last successdul reschedule. + * Delay the softlockup report when running a known slow code. + * It does _not_ affect the timestamp of the last successful reschedule. */ #define SOFTLOCKUP_DELAY_REPORT ULONG_MAX @@ -742,7 +742,7 @@ static int is_softlockup(unsigned long touch_ts, unsigned long period_ts, unsigned long now) { - if ((watchdog_enabled & WATCHDOG_SOFTOCKUP_ENABLED) && watchdog_thresh) { + if ((watchdog_enabled & WATCHDOG_SOFTLOCKUP_ENABLED) && watchdog_thresh) { /* * If period_ts has not been updated during a sample_period, then * in the subsequent few sample_periods, period_ts might also not @@ -1098,11 +1098,11 @@ static void proc_watchdog_update(bool thresh_changed) * caller | table->data points to | 'which' * -------------------|----------------------------------|------------------------------- * proc_watchdog | watchdog_user_enabled | WATCHDOG_HARDLOCKUP_ENABLED | - * | | WATCHDOG_SOFTOCKUP_ENABLED + * | | WATCHDOG_SOFTLOCKUP_ENABLED * -------------------|----------------------------------|------------------------------- * proc_nmi_watchdog | watchdog_hardlockup_user_enabled | WATCHDOG_HARDLOCKUP_ENABLED * -------------------|----------------------------------|------------------------------- - * proc_soft_watchdog | watchdog_softlockup_user_enabled | WATCHDOG_SOFTOCKUP_ENABLED + * proc_soft_watchdog | watchdog_softlockup_user_enabled | WATCHDOG_SOFTLOCKUP_ENABLED */ static int proc_watchdog_common(int which, const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) @@ -1136,7 +1136,7 @@ static int proc_watchdog(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { return proc_watchdog_common(WATCHDOG_HARDLOCKUP_ENABLED | - WATCHDOG_SOFTOCKUP_ENABLED, + WATCHDOG_SOFTLOCKUP_ENABLED, table, write, buffer, lenp, ppos); } @@ -1159,7 +1159,7 @@ static int proc_nmi_watchdog(const struct ctl_table *table, int write, static int proc_soft_watchdog(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { - return proc_watchdog_common(WATCHDOG_SOFTOCKUP_ENABLED, + return proc_watchdog_common(WATCHDOG_SOFTLOCKUP_ENABLED, table, write, buffer, lenp, ppos); } #endif diff --git a/lib/Kconfig b/lib/Kconfig index 55748b68714e..4e6b34c3346d 100644 --- a/lib/Kconfig +++ b/lib/Kconfig @@ -159,12 +159,6 @@ config AUDIT_COMPAT_GENERIC depends on AUDIT_GENERIC && AUDIT_ARCH_COMPAT_GENERIC && COMPAT default y -config RANDOM32_SELFTEST - bool "PRNG perform self test on init" - help - This option enables the 32 bit PRNG library functions to perform a - self test on initialization. - # # compression support is select'ed if needed # diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index e97bdf3a42a8..a2f0d3e97889 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -1292,10 +1292,10 @@ config DETECT_HUNG_TASK_BLOCKER depends on !PREEMPT_RT default y help - Say Y here to show the blocker task's stacktrace who acquires - the mutex lock which "hung tasks" are waiting. - This will add overhead a bit but shows suspicious tasks and - call trace if it comes from waiting a mutex. + Say Y here to show a possible blocker task's stacktrace when + a hung task is waiting on a mutex, semaphore, or rwsem. + This will add a bit of overhead but shows suspicious tasks and + call traces for waits on one of these locks. config WQ_WATCHDOG bool "Detect Workqueue Stalls" @@ -2219,10 +2219,11 @@ config KCOV_INSTRUMENT_ALL config KCOV_IRQ_AREA_SIZE hex "Size of interrupt coverage collection area in words" depends on KCOV + range 0x80 0x1000000 default 0x40000 help - KCOV uses preallocated per-cpu areas to collect coverage from - soft interrupts. This specifies the size of those areas in the + KCOV uses preallocated areas to collect coverage from soft + interrupts. This specifies the size of those areas in the number of unsigned long words. config KCOV_SELFTEST @@ -3489,6 +3490,23 @@ config GCD_KUNIT_TEST If unsure, say N +config POLYNOMIAL_KUNIT_TEST + tristate "Polynomial calculation (polynomial_calc) test" if !KUNIT_ALL_TESTS + depends on KUNIT + select POLYNOMIAL + default KUNIT_ALL_TESTS + help + This option enables the KUnit test suite for the polynomial_calc() + function, which evaluates integer polynomials using factor + redistribution to avoid overflow. + + The test suite verifies correctness for constant, linear, and + quadratic polynomials, negative coefficients, per-step dividers, + divider_leftover, total_divider scaling, and a real sensor + N-to-temperature conversion polynomial. + + If unsure, say N + config PRIME_NUMBERS_KUNIT_TEST tristate "Prime number generator test" if !KUNIT_ALL_TESTS depends on KUNIT @@ -3516,6 +3534,15 @@ config GLOB_KUNIT_TEST If unsure, say N +config PRANDOM_KUNIT_TEST + tristate "KUnit test for prandom" if !KUNIT_ALL_TESTS + depends on KUNIT + default KUNIT_ALL_TESTS + help + Enable this option to test the prandom functions at runtime. + + If unsure, say N + endif # RUNTIME_TESTING_MENU config ARCH_USE_MEMTEST diff --git a/lib/interval_tree_test.c b/lib/interval_tree_test.c index eba2d3e28980..b0b07270ce7c 100644 --- a/lib/interval_tree_test.c +++ b/lib/interval_tree_test.c @@ -140,13 +140,13 @@ static int intersection_range_check(void) intxn1 = bitmap_alloc(nnodes, GFP_KERNEL); if (!intxn1) { - WARN_ON_ONCE("Failed to allocate intxn1\n"); + WARN_ONCE(1, "Failed to allocate intxn1\n"); return -ENOMEM; } intxn2 = bitmap_alloc(nnodes, GFP_KERNEL); if (!intxn2) { - WARN_ON_ONCE("Failed to allocate intxn2\n"); + WARN_ONCE(1, "Failed to allocate intxn2\n"); bitmap_free(intxn1); return -ENOMEM; } diff --git a/lib/math/tests/Makefile b/lib/math/tests/Makefile index 13dc96e48408..85e1ad59f29d 100644 --- a/lib/math/tests/Makefile +++ b/lib/math/tests/Makefile @@ -4,5 +4,6 @@ obj-$(CONFIG_GCD_KUNIT_TEST) += gcd_kunit.o obj-$(CONFIG_INT_LOG_KUNIT_TEST) += int_log_kunit.o obj-$(CONFIG_INT_POW_KUNIT_TEST) += int_pow_kunit.o obj-$(CONFIG_INT_SQRT_KUNIT_TEST) += int_sqrt_kunit.o +obj-$(CONFIG_POLYNOMIAL_KUNIT_TEST) += polynomial_kunit.o obj-$(CONFIG_PRIME_NUMBERS_KUNIT_TEST) += prime_numbers_kunit.o obj-$(CONFIG_RATIONAL_KUNIT_TEST) += rational_kunit.o diff --git a/lib/math/tests/polynomial_kunit.c b/lib/math/tests/polynomial_kunit.c new file mode 100644 index 000000000000..ef443b57fc12 --- /dev/null +++ b/lib/math/tests/polynomial_kunit.c @@ -0,0 +1,270 @@ +// SPDX-License-Identifier: GPL-2.0-only + +#include <kunit/test.h> +#include <linux/polynomial.h> + +struct polynomial_test_param { + const struct polynomial *poly; + long data; + long expected; + const char *name; +}; + +/* f(x) = 5 */ +static const struct polynomial poly_constant = { + .total_divider = 1, + .terms = { + {0, 5, 1, 1}, + } +}; + +/* f(x) = 2x^2 + 3x + 5 */ +static const struct polynomial poly_simple = { + .total_divider = 1, + .terms = { + {2, 2, 1, 1}, + {1, 3, 1, 1}, + {0, 5, 1, 1}, + } +}; + +/* f(x) = -5x + 100 */ +static const struct polynomial poly_negative_coef = { + .total_divider = 1, + .terms = { + {1, -5, 1, 1}, + {0, 100, 1, 1}, + } +}; + +/* f(x) = (150x + 50) / 10 */ +static const struct polynomial poly_total_divider = { + .total_divider = 10, + .terms = { + {1, 150, 1, 1}, + {0, 50, 1, 1}, + } +}; + +/* + * f(x) = x / 2 + * divider=2 applied once per multiply: mult_frac(coef, data, 2) = coef*data/2 + */ +static const struct polynomial poly_step_divider = { + .total_divider = 1, + .terms = { + {1, 1, 2, 1}, + {0, 0, 1, 1}, + } +}; + +/* + * f(x) = (100/500) * x^2 = 0.2 * x^2 + * Encoded as coef=100, divider=10, divider_leftover=5: + * denom = 10^2 * 5 = 500 + */ +static const struct polynomial poly_leftover = { + .total_divider = 1, + .terms = { + {2, 100, 10, 5}, + {0, 0, 1, 1}, + } +}; + +/* + * f(x) = 2x^3 (single high-degree term, no constant) + * Used to exercise the power loop alone. + */ +static const struct polynomial poly_cubic = { + .total_divider = 1, + .terms = { + {3, 2, 1, 1}, + {0, 0, 1, 1}, + } +}; + +/* + * f(x) = 4x + 1 with a zero-coefficient quadratic term. + * The deg-2 term contributes nothing regardless of input. + */ +static const struct polynomial poly_zero_coef = { + .total_divider = 1, + .terms = { + {2, 0, 1, 1}, + {1, 4, 1, 1}, + {0, 1, 1, 1}, + } +}; + +/* + * f(x) = 9 with total_divider = 0. + * The implementation treats 0 as 1 via `total_divider ?: 1`, so the + * result must equal the constant term unchanged. + */ +static const struct polynomial poly_zero_total_divider = { + .total_divider = 0, + .terms = { + {0, 9, 1, 1}, + } +}; + + +static const struct polynomial_test_param test_params[] = { + { + .poly = &poly_constant, + .data = 0, + .expected = 5, + .name = "Constant polynomial at x=0", + }, + { + .poly = &poly_constant, + .data = 42, + .expected = 5, + .name = "Constant polynomial is independent of input", + }, + { + .poly = &poly_simple, + .data = 0, + .expected = 5, /* zero input collapses all power terms */ + .name = "Zero input yields constant term only", + }, + { + .poly = &poly_simple, + .data = 10, + .expected = 235, /* 2*100 + 3*10 + 5 */ + .name = "Simple quadratic at x=10", + }, + { + .poly = &poly_negative_coef, + .data = 10, + .expected = 50, /* -5*10 + 100 */ + .name = "Negative coefficient at x=10", + }, + { + .poly = &poly_negative_coef, + .data = 20, + .expected = 0, /* -5*20 + 100 = 0 */ + .name = "Negative coefficient result is zero", + }, + { + .poly = &poly_total_divider, + .data = 3, + .expected = 50, /* (150*3 + 50) / 10 = 500/10 */ + .name = "total_divider scales the final sum", + }, + { + .poly = &poly_step_divider, + .data = 100, + .expected = 50, /* 1*100/2 */ + .name = "Per-step divider halves input", + }, + { + .poly = &poly_leftover, + .data = 30, + .expected = 180, /* 100*30^2 / (10^2 * 5) = 90000/500 */ + .name = "divider_leftover with quadratic term", + }, + /* Boundary: unit and negative-unit input */ + { + /* + * data=1: each mult_frac(tmp, 1, divider) strips one factor of + * divider from coef per degree, so coef is left-shifted right + * until intermediate precision is exhausted. + * 2*1 + 3*1 + 5 = 10 + */ + .poly = &poly_simple, + .data = 1, + .expected = 10, + .name = "Boundary: data=1 (unit input)", + }, + { + /* + * data=-1: even degrees produce positive contributions, + * odd degrees produce negative ones. + * 2*(-1)^2 + 3*(-1) + 5 = 2 - 3 + 5 = 4 + */ + .poly = &poly_simple, + .data = -1, + .expected = 4, + .name = "Boundary: data=-1 (negative unit input)", + }, + + /* Boundary: negative non-trivial input */ + { + /* + * 2*(-3)^2 + 3*(-3) + 5 = 18 - 9 + 5 = 14 + * Verifies sign handling for negative data across all degrees. + */ + .poly = &poly_simple, + .data = -3, + .expected = 14, + .name = "Boundary: negative data with quadratic", + }, + + /* Boundary: total_divider = 0 is treated as 1 */ + { + .poly = &poly_zero_total_divider, + .data = 42, + .expected = 9, + .name = "Boundary: total_divider=0 defaults to 1", + }, + + /* Boundary: zero-coefficient high-degree term */ + { + /* + * The deg-2 term has coef=0, so it contributes 0 regardless + * of data. Result: 0 + 4*10 + 1 = 41 + */ + .poly = &poly_zero_coef, + .data = 10, + .expected = 41, + .name = "Boundary: zero-coefficient term is inert", + }, + + /* Boundary: single high-degree term, no constant */ + { + /* 2 * 5^3 = 250; also verifies the loop terminates on deg-0 */ + .poly = &poly_cubic, + .data = 5, + .expected = 250, + .name = "Boundary: single cubic term", + }, + { + /* 2 * (-2)^3 = -16; odd power preserves sign of negative data */ + .poly = &poly_cubic, + .data = -2, + .expected = -16, + .name = "Boundary: single cubic term, negative data", + }, + +}; + +static void get_desc(const struct polynomial_test_param *param, char *desc) +{ + strscpy(desc, param->name, KUNIT_PARAM_DESC_SIZE); +} + +KUNIT_ARRAY_PARAM(polynomial, test_params, get_desc); + +static void polynomial_calc_test(struct kunit *test) +{ + const struct polynomial_test_param *param = test->param_value; + + KUNIT_EXPECT_EQ(test, polynomial_calc(param->poly, param->data), + param->expected); +} + +static struct kunit_case polynomial_test_cases[] = { + KUNIT_CASE_PARAM(polynomial_calc_test, polynomial_gen_params), + {} +}; + +static struct kunit_suite polynomial_test_suite = { + .name = "math-polynomial", + .test_cases = polynomial_test_cases, +}; + +kunit_test_suites(&polynomial_test_suite); + +MODULE_DESCRIPTION("math.polynomial_calc KUnit test suite"); +MODULE_LICENSE("GPL"); diff --git a/lib/raid/Kconfig b/lib/raid/Kconfig index 978cd6ba08ac..01f007b2522c 100644 --- a/lib/raid/Kconfig +++ b/lib/raid/Kconfig @@ -29,6 +29,12 @@ config XOR_KUNIT_TEST This is intended to help people writing architecture-specific optimized versions. If unsure, say N. +config XOR_BENCHMARK + bool "Benchmark for xor_gen" + depends on XOR_KUNIT_TEST + help + Include benchmarks in the KUnit test suite for xor_gen. + config RAID6_PQ tristate @@ -54,6 +60,12 @@ config RAID6_PQ_KUNIT_TEST This is intended to help people writing architecture-specific optimized versions. If unsure, say N. +config RAID6_PQ_KUNIT_BENCHMARK + bool "Benchmark for RAID6 PQ" + depends on RAID6_PQ_KUNIT_TEST + help + Include benchmarks in the KUnit test suite for raid P/Q generation. + config RAID6_PQ_BENCHMARK bool "Automatically choose fastest RAID6 PQ functions" depends on RAID6_PQ diff --git a/lib/raid/raid6/Makefile b/lib/raid/raid6/Makefile index 038d6c74d1ba..5cdb7223de2a 100644 --- a/lib/raid/raid6/Makefile +++ b/lib/raid/raid6/Makefile @@ -1,5 +1,7 @@ # SPDX-License-Identifier: GPL-2.0 +CONTEXT_ANALYSIS := y + ccflags-y += -I $(src) ifeq ($(CONFIG_RAID6_PQ_ARCH),y) diff --git a/lib/raid/raid6/algos.c b/lib/raid/raid6/algos.c index 6f5c89ab2b17..011aa9d0eb3c 100644 --- a/lib/raid/raid6/algos.c +++ b/lib/raid/raid6/algos.c @@ -152,40 +152,37 @@ void raid6_recov_datap(int disks, size_t bytes, int faila, void **ptrs) } EXPORT_SYMBOL_GPL(raid6_recov_datap); -#define RAID6_TIME_JIFFIES_LG2 4 -#define RAID6_TEST_DISKS 8 +#define BENCH_SIZE SZ_4K +#define NR_SRCS 8 +#define NR_DISKS (NR_SRCS + 2) +#define REPS 800U -static int raid6_choose_gen(void *(*const dptrs)[RAID6_TEST_DISKS], - const int disks) +static int raid6_choose_gen(void *dptrs[NR_DISKS], const int disks) { - /* work on the second half of the disks */ - int start = (disks >> 1) - 1, stop = disks - 3; const struct raid6_calls *best = NULL; unsigned long bestgenperf = 0; unsigned int i; for (i = 0; i < raid6_nr_algos; i++) { const struct raid6_calls *algo = raid6_algos[i]; - unsigned long perf = 0, j0, j1; + unsigned long perf = 0; + u64 t; + int i; preempt_disable(); - j0 = jiffies; - while ((j1 = jiffies) == j0) - cpu_relax(); - while (time_before(jiffies, - j1 + (1<<RAID6_TIME_JIFFIES_LG2))) { - algo->gen_syndrome(disks, PAGE_SIZE, *dptrs); - perf++; - } + t = ktime_get_ns(); + for (i = 0; i < REPS; i++) + algo->gen_syndrome(disks, BENCH_SIZE, dptrs); + t = max(ktime_get_ns() - t, 1); preempt_enable(); + /* bytes/ns == GB/s, multiply by 1000 to get MB/s [not MiB/s] */ + perf = div64_u64((u64)BENCH_SIZE * REPS * NR_SRCS * 1000, t); if (perf > bestgenperf) { bestgenperf = perf; best = algo; } - pr_info("raid6: %-8s gen() %5ld MB/s\n", algo->name, - (perf * HZ * (disks-2)) >> - (20 - PAGE_SHIFT + RAID6_TIME_JIFFIES_LG2)); + pr_info("raid6: %-8s gen() %5lu MB/s\n", algo->name, perf); } if (!best) { @@ -197,28 +194,24 @@ static int raid6_choose_gen(void *(*const dptrs)[RAID6_TEST_DISKS], static_call_update(raid6_xor_syndrome_impl, best->xor_syndrome); pr_info("raid6: using algorithm %s gen() %ld MB/s\n", - best->name, - (bestgenperf * HZ * (disks - 2)) >> - (20 - PAGE_SHIFT + RAID6_TIME_JIFFIES_LG2)); + best->name, bestgenperf); if (best->xor_syndrome) { - unsigned long perf = 0, j0, j1; + /* work on the second half of the disks */ + int start = (disks / 2) - 1, stop = disks - 3; + u64 t; preempt_disable(); - j0 = jiffies; - while ((j1 = jiffies) == j0) - cpu_relax(); - while (time_before(jiffies, - j1 + (1 << RAID6_TIME_JIFFIES_LG2))) { - best->xor_syndrome(disks, start, stop, - PAGE_SIZE, *dptrs); - perf++; - } + t = ktime_get_ns(); + for (i = 0; i < REPS; i++) + best->xor_syndrome(disks, start, stop, BENCH_SIZE, + dptrs); + t = max(ktime_get_ns() - t, 1); preempt_enable(); - pr_info("raid6: .... xor() %ld MB/s, rmw enabled\n", - (perf * HZ * (disks - 2)) >> - (20 - PAGE_SHIFT + RAID6_TIME_JIFFIES_LG2 + 1)); + pr_info("raid6: .... xor() %llu MB/s, rmw enabled\n", + div64_u64((u64)BENCH_SIZE * REPS * NR_SRCS / 2 * 1000, + t)); } return 0; @@ -230,24 +223,20 @@ static int raid6_choose_gen(void *(*const dptrs)[RAID6_TEST_DISKS], static int __init raid6_select_algo(void) { - const int disks = RAID6_TEST_DISKS; + const int disks = NR_DISKS; + void *dptrs[NR_DISKS]; char *disk_ptr, *p; - void *dptrs[RAID6_TEST_DISKS]; int i, cycle; int error; if (!IS_ENABLED(CONFIG_RAID6_PQ_BENCHMARK) || raid6_nr_algos == 1) { pr_info("raid6: skipped pq benchmark and selected %s\n", raid6_algos[raid6_nr_algos - 1]->name); - static_call_update(raid6_gen_syndrome_impl, - raid6_algos[raid6_nr_algos - 1]->gen_syndrome); - static_call_update(raid6_xor_syndrome_impl, - raid6_algos[raid6_nr_algos - 1]->xor_syndrome); return 0; } /* prepare the buffer and fill it circularly with gfmul table */ - disk_ptr = kmalloc(PAGE_SIZE * RAID6_TEST_DISKS, GFP_KERNEL); + disk_ptr = kmalloc_array(NR_DISKS, BENCH_SIZE, GFP_KERNEL); if (!disk_ptr) { pr_err("raid6: Yikes! No memory available.\n"); return -ENOMEM; @@ -255,19 +244,19 @@ static int __init raid6_select_algo(void) p = disk_ptr; for (i = 0; i < disks; i++) - dptrs[i] = p + PAGE_SIZE * i; + dptrs[i] = p + BENCH_SIZE * i; - cycle = ((disks - 2) * PAGE_SIZE) / 65536; + cycle = ((disks - 2) * BENCH_SIZE) / 65536; for (i = 0; i < cycle; i++) { memcpy(p, raid6_gfmul, 65536); p += 65536; } - if ((disks - 2) * PAGE_SIZE % 65536) - memcpy(p, raid6_gfmul, (disks - 2) * PAGE_SIZE % 65536); + if ((disks - 2) * BENCH_SIZE % 65536) + memcpy(p, raid6_gfmul, (disks - 2) * BENCH_SIZE % 65536); /* select raid gen_syndrome function */ - error = raid6_choose_gen(&dptrs, disks); + error = raid6_choose_gen(dptrs, disks); kfree(disk_ptr); @@ -329,13 +318,34 @@ static int __init raid6_init(void) static_call_update(raid6_recov_datap_impl, raid6_recov_algo->datap); pr_info("raid6: using %s recovery algorithm\n", raid6_recov_algo->name); + /* + * Pick the last registered implementation as the temporary default until + * calibration happens. + */ + static_call_update(raid6_gen_syndrome_impl, + raid6_algos[raid6_nr_algos - 1]->gen_syndrome); + static_call_update(raid6_xor_syndrome_impl, + raid6_algos[raid6_nr_algos - 1]->xor_syndrome); + +#ifdef MODULE return raid6_select_algo(); +#else + return 0; +#endif } static void __exit raid6_exit(void) { } +/* + * When built-in we must register the default implementation before md + * initializes, but we don't want calibration to run that early as that + * would delay the boot process. + */ +#ifndef MODULE +device_initcall(raid6_select_algo); +#endif subsys_initcall(raid6_init); module_exit(raid6_exit); MODULE_LICENSE("GPL"); diff --git a/lib/raid/raid6/tests/raid6_kunit.c b/lib/raid/raid6/tests/raid6_kunit.c index 9f3e671a1224..7d0a1ec98001 100644 --- a/lib/raid/raid6/tests/raid6_kunit.c +++ b/lib/raid/raid6/tests/raid6_kunit.c @@ -18,6 +18,7 @@ MODULE_IMPORT_NS("EXPORTED_FOR_KUNIT_TESTING"); #define RAID6_KUNIT_MAX_BUFFERS 64 /* Including P and Q */ #define RAID6_KUNIT_MAX_FAILURES 2 #define RAID6_KUNIT_MAX_BYTES PAGE_SIZE +#define RAID6_KUNIT_ALLOC_BYTES SZ_16K static struct rnd_state rng; static void *test_buffers[RAID6_KUNIT_MAX_BUFFERS]; @@ -229,6 +230,68 @@ static void raid6_test(struct kunit *test) raid6_test_one(test); } +static void raid6_benchmark(struct kunit *test) +{ + static const unsigned int nr_to_test[] = { + 4, 5, 6, 7, 8, 10, 12, 15, 16, 32, + }; + static const unsigned int len_to_test[] = { + SZ_4K, SZ_16K, + }; + unsigned int i, j, l; + u64 t; + + if (!IS_ENABLED(CONFIG_RAID6_PQ_KUNIT_BENCHMARK)) + kunit_skip(test, "not enabled"); + + /* warm-up */ + for (i = 0; i < ARRAY_SIZE(nr_to_test); i++) { + for (j = 0; j < ARRAY_SIZE(len_to_test); j++) { + for (l = 0; l < 10; l++) { + raid6_gen_syndrome(nr_to_test[i], + len_to_test[j], test_buffers); + } + } + } + + /* + * Preferably this would be a loop over len_to_test, but the kunit + * logging always adds a newline to each logged format string. + */ + static_assert(ARRAY_SIZE(len_to_test) == 2); + kunit_info(test, " \t%5u bytes\t%5u bytes\n", + len_to_test[0], len_to_test[1]); + + for (i = 0; i < ARRAY_SIZE(nr_to_test); i++) { + unsigned int nr = nr_to_test[i]; + u64 speed[ARRAY_SIZE(len_to_test)]; + + KUNIT_ASSERT_LE(test, nr, RAID6_KUNIT_MAX_BUFFERS); + + for (j = 0; j < ARRAY_SIZE(len_to_test); j++) { + unsigned int len = len_to_test[j]; + const unsigned long num_iters = 1000; + + KUNIT_ASSERT_GT(test, len, 0); + KUNIT_ASSERT_LE(test, len, RAID6_KUNIT_ALLOC_BYTES); + + preempt_disable(); + t = ktime_get_ns(); + for (l = 0; l < num_iters; l++) + raid6_gen_syndrome(nr_to_test[i], + len_to_test[j], test_buffers); + t = max(ktime_get_ns() - t, 1); + preempt_enable(); + + speed[j] = div64_u64((u64)len * num_iters * nr, t); + } + + static_assert(ARRAY_SIZE(len_to_test) == 2); + kunit_info(test, "%3u disks:\t%5llu GB/s\t%5llu GB/s\n", + nr, speed[0], speed[1]); + } +} + static const void *raid6_gen_params(struct kunit *test, const void *prev, char *desc) { @@ -256,6 +319,7 @@ next_algo: static struct kunit_case raid6_test_cases[] = { KUNIT_CASE_PARAM(raid6_test, raid6_gen_params), + KUNIT_CASE(raid6_benchmark), {}, }; @@ -270,7 +334,7 @@ static int raid6_suite_init(struct kunit_suite *suite) * so that it is immediately followed by a guard page. This allows * buffer overreads to be detected, even in assembly code. */ - test_buflen = round_up(RAID6_KUNIT_MAX_BYTES, PAGE_SIZE); + test_buflen = round_up(RAID6_KUNIT_ALLOC_BYTES, PAGE_SIZE); for (i = 0; i < RAID6_KUNIT_MAX_FAILURES; i++) { test_recov_buffers[i] = vmalloc(test_buflen); if (!test_recov_buffers[i]) diff --git a/lib/raid/xor/Makefile b/lib/raid/xor/Makefile index e8ecec3c09f9..9b0fad459cdb 100644 --- a/lib/raid/xor/Makefile +++ b/lib/raid/xor/Makefile @@ -1,5 +1,7 @@ # SPDX-License-Identifier: GPL-2.0 +CONTEXT_ANALYSIS := y + ccflags-y += -I $(src) obj-$(CONFIG_XOR_BLOCKS) += xor.o diff --git a/lib/raid/xor/tests/Makefile b/lib/raid/xor/tests/Makefile index 661e8f6ffd1f..1cce833cd7fd 100644 --- a/lib/raid/xor/tests/Makefile +++ b/lib/raid/xor/tests/Makefile @@ -1,3 +1,5 @@ # SPDX-License-Identifier: GPL-2.0-only +CONTEXT_ANALYSIS := y + obj-$(CONFIG_XOR_KUNIT_TEST) += xor_kunit.o diff --git a/lib/raid/xor/tests/xor_kunit.c b/lib/raid/xor/tests/xor_kunit.c index 0c2a3a420bf9..1fb30b2c7c47 100644 --- a/lib/raid/xor/tests/xor_kunit.c +++ b/lib/raid/xor/tests/xor_kunit.c @@ -85,7 +85,7 @@ static void xor_test(struct kunit *test) xor_generate_random_data(); /* - * If we're not using the entire buffer size, inject randomize + * If we're not using the entire buffer size, inject randomized * alignment into the buffer. */ max_alignment = XOR_KUNIT_MAX_BYTES - len; @@ -125,8 +125,70 @@ static void xor_test(struct kunit *test) } } +static void xor_benchmark(struct kunit *test) +{ + static const unsigned int nr_to_test[] = { + 4, 5, 6, 7, 8, 10, 12, 15, 16, 32, + }; + static const unsigned int len_to_test[] = { + SZ_4K, SZ_16K, + }; + unsigned int i, j, l; + u64 t; + + if (!IS_ENABLED(CONFIG_XOR_BENCHMARK)) + kunit_skip(test, "not enabled"); + + /* warm-up */ + for (i = 0; i < ARRAY_SIZE(nr_to_test); i++) { + for (j = 0; j < ARRAY_SIZE(len_to_test); j++) { + for (l = 0; l < 10; l++) { + xor_gen(test_dest, test_buffers, nr_to_test[i], + len_to_test[j]); + } + } + } + + /* + * Preferably this would be a loop over len_to_test, but the kunit + * logging always adds a newline to each logged format string. + */ + static_assert(ARRAY_SIZE(len_to_test) == 2); + kunit_info(test, " \t%5u bytes\t%5u bytes\n", + len_to_test[0], len_to_test[1]); + + for (i = 0; i < ARRAY_SIZE(nr_to_test); i++) { + unsigned int nr = nr_to_test[i]; + u64 speed[ARRAY_SIZE(len_to_test)]; + + KUNIT_ASSERT_LE(test, nr, XOR_KUNIT_MAX_BUFFERS); + + for (j = 0; j < ARRAY_SIZE(len_to_test); j++) { + unsigned int len = len_to_test[j]; + const unsigned long num_iters = 1000; + + KUNIT_ASSERT_GT(test, len, 0); + KUNIT_ASSERT_LE(test, len, XOR_KUNIT_MAX_BYTES); + + preempt_disable(); + t = ktime_get_ns(); + for (l = 0; l < num_iters; l++) + xor_gen(test_dest, test_buffers, nr, len); + t = max(ktime_get_ns() - t, 1); + preempt_enable(); + + speed[j] = div64_u64((u64)len * num_iters * nr, t); + } + + static_assert(ARRAY_SIZE(len_to_test) == 2); + kunit_info(test, "%3u disks:\t%5llu GB/s\t%5llu GB/s\n", + nr, speed[0], speed[1]); + } +} + static struct kunit_case xor_test_cases[] = { KUNIT_CASE(xor_test), + KUNIT_CASE(xor_benchmark), {}, }; diff --git a/lib/raid/xor/xor-core.c b/lib/raid/xor/xor-core.c index 50931fbf0324..f9d61b096b11 100644 --- a/lib/raid/xor/xor-core.c +++ b/lib/raid/xor/xor-core.c @@ -10,7 +10,6 @@ #include <linux/gfp.h> #include <linux/slab.h> #include <linux/raid/xor.h> -#include <linux/jiffies.h> #include <linux/preempt.h> #include <linux/static_call.h> #include "xor_impl.h" @@ -73,59 +72,56 @@ void __init xor_force(struct xor_block_template *tmpl) forced_template = tmpl; } -#define BENCH_SIZE 4096 +#define BENCH_SIZE SZ_4K +#define NR_SRCS 4 #define REPS 800U -static void __init -do_xor_speed(struct xor_block_template *tmpl, void *b1, void *b2) +static void __init do_xor_speed(struct xor_block_template *tmpl, void *dest, + void *srcs[NR_SRCS]) { - int speed; - unsigned long reps; - ktime_t min, start, t0; - void *srcs[1] = { b2 }; + u64 t; + int i; preempt_disable(); - - reps = 0; - t0 = ktime_get(); - /* delay start until time has advanced */ - while ((start = ktime_get()) == t0) - cpu_relax(); - do { + t = ktime_get_ns(); + for (i = 0; i < REPS; i++) { mb(); /* prevent loop optimization */ - tmpl->xor_gen(b1, srcs, 1, BENCH_SIZE); + tmpl->xor_gen(dest, srcs, NR_SRCS, BENCH_SIZE); mb(); - } while (reps++ < REPS || (t0 = ktime_get()) == start); - min = ktime_sub(t0, start); - + } + t = max(ktime_get_ns() - t, 1); preempt_enable(); - // bytes/ns == GB/s, multiply by 1000 to get MB/s [not MiB/s] - speed = (1000 * reps * BENCH_SIZE) / (unsigned int)ktime_to_ns(min); - tmpl->speed = speed; + /* bytes/ns == GB/s, multiply by 1000 to get MB/s [not MiB/s] */ + tmpl->speed = div64_u64((u64)BENCH_SIZE * REPS * NR_SRCS * 1000, t); - pr_info(" %-16s: %5d MB/sec\n", tmpl->name, speed); + pr_info(" %-16s: %5d MB/sec\n", tmpl->name, tmpl->speed); } static int __init calibrate_xor_blocks(void) { - void *b1, *b2; struct xor_block_template *f, *fastest; + void *srcs[NR_SRCS]; + void *buf, *dest; + int i; if (forced_template) return 0; - b1 = kmalloc(PAGE_SIZE * 4, GFP_KERNEL); - if (!b1) { + buf = kmalloc(BENCH_SIZE * (NR_SRCS + 1), GFP_KERNEL); + if (!buf) { pr_warn("xor: Yikes! No memory available.\n"); return -ENOMEM; } - b2 = b1 + 2*PAGE_SIZE + BENCH_SIZE; + get_random_bytes(buf, BENCH_SIZE * (NR_SRCS + 1)); + dest = buf; + for (i = 0; i < NR_SRCS; i++) + srcs[i] = buf + (i + 1) * BENCH_SIZE; pr_info("xor: measuring software checksum speed\n"); fastest = template_list; for (f = template_list; f; f = f->next) { - do_xor_speed(f, b1, b2); + do_xor_speed(f, dest, srcs); if (f->speed > fastest->speed) fastest = f; } @@ -133,9 +129,10 @@ static int __init calibrate_xor_blocks(void) pr_info("xor: using function: %s (%d MB/sec)\n", fastest->name, fastest->speed); - kfree(b1); + kfree(buf); return 0; } +#undef NR_SRCS #ifdef CONFIG_XOR_BLOCKS_ARCH #include "xor_arch.h" /* $SRCARCH/xor_arch.h */ diff --git a/lib/random32.c b/lib/random32.c index 24e7acd9343f..dad90219c351 100644 --- a/lib/random32.c +++ b/lib/random32.c @@ -41,6 +41,7 @@ #include <linux/bitops.h> #include <linux/slab.h> #include <linux/unaligned.h> +#include <kunit/visibility.h> /** * prandom_u32_state - seeded pseudo-random number generator. @@ -92,7 +93,14 @@ void prandom_bytes_state(struct rnd_state *state, void *buf, size_t bytes) } EXPORT_SYMBOL(prandom_bytes_state); -static void prandom_warmup(struct rnd_state *state) +/* + * Only declared here so that it has a prototype when made + * non-static for KUnit testing (avoids -Wmissing-prototypes). + */ +#if IS_ENABLED(CONFIG_KUNIT) +void prandom_warmup(struct rnd_state *state); +#endif +VISIBLE_IF_KUNIT void prandom_warmup(struct rnd_state *state) { /* Calling RNG ten times to satisfy recurrence condition */ prandom_u32_state(state); @@ -106,6 +114,7 @@ static void prandom_warmup(struct rnd_state *state) prandom_u32_state(state); prandom_u32_state(state); } +EXPORT_SYMBOL_IF_KUNIT(prandom_warmup); void prandom_seed_full_state(struct rnd_state __percpu *pcpu_state) { @@ -125,176 +134,3 @@ void prandom_seed_full_state(struct rnd_state __percpu *pcpu_state) } } EXPORT_SYMBOL(prandom_seed_full_state); - -#ifdef CONFIG_RANDOM32_SELFTEST -static struct prandom_test1 { - u32 seed; - u32 result; -} test1[] = { - { 1U, 3484351685U }, - { 2U, 2623130059U }, - { 3U, 3125133893U }, - { 4U, 984847254U }, -}; - -static struct prandom_test2 { - u32 seed; - u32 iteration; - u32 result; -} test2[] = { - /* Test cases against taus113 from GSL library. */ - { 931557656U, 959U, 2975593782U }, - { 1339693295U, 876U, 3887776532U }, - { 1545556285U, 961U, 1615538833U }, - { 601730776U, 723U, 1776162651U }, - { 1027516047U, 687U, 511983079U }, - { 416526298U, 700U, 916156552U }, - { 1395522032U, 652U, 2222063676U }, - { 366221443U, 617U, 2992857763U }, - { 1539836965U, 714U, 3783265725U }, - { 556206671U, 994U, 799626459U }, - { 684907218U, 799U, 367789491U }, - { 2121230701U, 931U, 2115467001U }, - { 1668516451U, 644U, 3620590685U }, - { 768046066U, 883U, 2034077390U }, - { 1989159136U, 833U, 1195767305U }, - { 536585145U, 996U, 3577259204U }, - { 1008129373U, 642U, 1478080776U }, - { 1740775604U, 939U, 1264980372U }, - { 1967883163U, 508U, 10734624U }, - { 1923019697U, 730U, 3821419629U }, - { 442079932U, 560U, 3440032343U }, - { 1961302714U, 845U, 841962572U }, - { 2030205964U, 962U, 1325144227U }, - { 1160407529U, 507U, 240940858U }, - { 635482502U, 779U, 4200489746U }, - { 1252788931U, 699U, 867195434U }, - { 1961817131U, 719U, 668237657U }, - { 1071468216U, 983U, 917876630U }, - { 1281848367U, 932U, 1003100039U }, - { 582537119U, 780U, 1127273778U }, - { 1973672777U, 853U, 1071368872U }, - { 1896756996U, 762U, 1127851055U }, - { 847917054U, 500U, 1717499075U }, - { 1240520510U, 951U, 2849576657U }, - { 1685071682U, 567U, 1961810396U }, - { 1516232129U, 557U, 3173877U }, - { 1208118903U, 612U, 1613145022U }, - { 1817269927U, 693U, 4279122573U }, - { 1510091701U, 717U, 638191229U }, - { 365916850U, 807U, 600424314U }, - { 399324359U, 702U, 1803598116U }, - { 1318480274U, 779U, 2074237022U }, - { 697758115U, 840U, 1483639402U }, - { 1696507773U, 840U, 577415447U }, - { 2081979121U, 981U, 3041486449U }, - { 955646687U, 742U, 3846494357U }, - { 1250683506U, 749U, 836419859U }, - { 595003102U, 534U, 366794109U }, - { 47485338U, 558U, 3521120834U }, - { 619433479U, 610U, 3991783875U }, - { 704096520U, 518U, 4139493852U }, - { 1712224984U, 606U, 2393312003U }, - { 1318233152U, 922U, 3880361134U }, - { 855572992U, 761U, 1472974787U }, - { 64721421U, 703U, 683860550U }, - { 678931758U, 840U, 380616043U }, - { 692711973U, 778U, 1382361947U }, - { 677703619U, 530U, 2826914161U }, - { 92393223U, 586U, 1522128471U }, - { 1222592920U, 743U, 3466726667U }, - { 358288986U, 695U, 1091956998U }, - { 1935056945U, 958U, 514864477U }, - { 735675993U, 990U, 1294239989U }, - { 1560089402U, 897U, 2238551287U }, - { 70616361U, 829U, 22483098U }, - { 368234700U, 731U, 2913875084U }, - { 20221190U, 879U, 1564152970U }, - { 539444654U, 682U, 1835141259U }, - { 1314987297U, 840U, 1801114136U }, - { 2019295544U, 645U, 3286438930U }, - { 469023838U, 716U, 1637918202U }, - { 1843754496U, 653U, 2562092152U }, - { 400672036U, 809U, 4264212785U }, - { 404722249U, 965U, 2704116999U }, - { 600702209U, 758U, 584979986U }, - { 519953954U, 667U, 2574436237U }, - { 1658071126U, 694U, 2214569490U }, - { 420480037U, 749U, 3430010866U }, - { 690103647U, 969U, 3700758083U }, - { 1029424799U, 937U, 3787746841U }, - { 2012608669U, 506U, 3362628973U }, - { 1535432887U, 998U, 42610943U }, - { 1330635533U, 857U, 3040806504U }, - { 1223800550U, 539U, 3954229517U }, - { 1322411537U, 680U, 3223250324U }, - { 1877847898U, 945U, 2915147143U }, - { 1646356099U, 874U, 965988280U }, - { 805687536U, 744U, 4032277920U }, - { 1948093210U, 633U, 1346597684U }, - { 392609744U, 783U, 1636083295U }, - { 690241304U, 770U, 1201031298U }, - { 1360302965U, 696U, 1665394461U }, - { 1220090946U, 780U, 1316922812U }, - { 447092251U, 500U, 3438743375U }, - { 1613868791U, 592U, 828546883U }, - { 523430951U, 548U, 2552392304U }, - { 726692899U, 810U, 1656872867U }, - { 1364340021U, 836U, 3710513486U }, - { 1986257729U, 931U, 935013962U }, - { 407983964U, 921U, 728767059U }, -}; - -static void prandom_state_selftest_seed(struct rnd_state *state, u32 seed) -{ -#define LCG(x) ((x) * 69069U) /* super-duper LCG */ - state->s1 = __seed(LCG(seed), 2U); - state->s2 = __seed(LCG(state->s1), 8U); - state->s3 = __seed(LCG(state->s2), 16U); - state->s4 = __seed(LCG(state->s3), 128U); -} - -static int __init prandom_state_selftest(void) -{ - int i, j, errors = 0, runs = 0; - bool error = false; - - for (i = 0; i < ARRAY_SIZE(test1); i++) { - struct rnd_state state; - - prandom_state_selftest_seed(&state, test1[i].seed); - prandom_warmup(&state); - - if (test1[i].result != prandom_u32_state(&state)) - error = true; - } - - if (error) - pr_warn("prandom: seed boundary self test failed\n"); - else - pr_info("prandom: seed boundary self test passed\n"); - - for (i = 0; i < ARRAY_SIZE(test2); i++) { - struct rnd_state state; - - prandom_state_selftest_seed(&state, test2[i].seed); - prandom_warmup(&state); - - for (j = 0; j < test2[i].iteration - 1; j++) - prandom_u32_state(&state); - - if (test2[i].result != prandom_u32_state(&state)) - errors++; - - runs++; - cond_resched(); - } - - if (errors) - pr_warn("prandom: %d/%d self tests failed\n", errors, runs); - else - pr_info("prandom: %d self tests passed\n", runs); - return 0; -} -core_initcall(prandom_state_selftest); -#endif diff --git a/lib/string.c b/lib/string.c index a4e8ad23577d..a3778d5aab4a 100644 --- a/lib/string.c +++ b/lib/string.c @@ -821,7 +821,8 @@ void *memchr_inv(const void *start, int c, size_t bytes) { u8 value = c; u64 value64; - unsigned int words, prefix; + size_t words; + unsigned int prefix; if (bytes <= 16) return check_bytes8(start, value, bytes); diff --git a/lib/test_hmm.c b/lib/test_hmm.c index 6205fb313bd0..6e72a0d8a53d 100644 --- a/lib/test_hmm.c +++ b/lib/test_hmm.c @@ -1312,16 +1312,10 @@ static int dmirror_migrate_to_device(struct dmirror *dmirror, if (!mmget_not_zero(mm)) return -EINVAL; - ret = -ENOMEM; src_pfns = kvcalloc(PTRS_PER_PTE, sizeof(*src_pfns), GFP_KERNEL | __GFP_NOFAIL); - if (!src_pfns) - goto free_mem; - dst_pfns = kvcalloc(PTRS_PER_PTE, sizeof(*dst_pfns), GFP_KERNEL | __GFP_NOFAIL); - if (!dst_pfns) - goto free_mem; ret = 0; mmap_read_lock(mm); diff --git a/lib/tests/Makefile b/lib/tests/Makefile index 4ead57602eac..3cac3b63a752 100644 --- a/lib/tests/Makefile +++ b/lib/tests/Makefile @@ -41,8 +41,7 @@ obj-$(CONFIG_MEMCPY_KUNIT_TEST) += memcpy_kunit.o obj-$(CONFIG_MIN_HEAP_KUNIT_TEST) += min_heap_kunit.o CFLAGS_overflow_kunit.o = $(call cc-disable-warning, tautological-constant-out-of-range-compare) obj-$(CONFIG_OVERFLOW_KUNIT_TEST) += overflow_kunit.o -# GCC < 12.1 can miscompile errptr() test when branch profiling is enabled. -CFLAGS_printf_kunit.o += -DDISABLE_BRANCH_PROFILING +obj-$(CONFIG_PRANDOM_KUNIT_TEST) += random32_kunit.o obj-$(CONFIG_PRINTF_KUNIT_TEST) += printf_kunit.o obj-$(CONFIG_RANDSTRUCT_KUNIT_TEST) += randstruct_kunit.o obj-$(CONFIG_SCANF_KUNIT_TEST) += scanf_kunit.o diff --git a/lib/tests/random32_kunit.c b/lib/tests/random32_kunit.c new file mode 100644 index 000000000000..0b4af2b09c01 --- /dev/null +++ b/lib/tests/random32_kunit.c @@ -0,0 +1,182 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Test cases for random32 functions. + */ + +#include <linux/prandom.h> +#include <kunit/test.h> + +/* prandom_warmup() is static in lib/random32.c; exposed for testing only. */ +void prandom_warmup(struct rnd_state *state); + +static const struct prandom_test1 { + u32 seed; + u32 result; +} test1[] = { + { 1U, 3484351685U }, + { 2U, 2623130059U }, + { 3U, 3125133893U }, + { 4U, 984847254U }, +}; + +static const struct prandom_test2 { + u32 seed; + u32 iteration; + u32 result; +} test2[] = { + /* Test cases against taus113 from GSL library. */ + { 931557656U, 959U, 2975593782U }, + { 1339693295U, 876U, 3887776532U }, + { 1545556285U, 961U, 1615538833U }, + { 601730776U, 723U, 1776162651U }, + { 1027516047U, 687U, 511983079U }, + { 416526298U, 700U, 916156552U }, + { 1395522032U, 652U, 2222063676U }, + { 366221443U, 617U, 2992857763U }, + { 1539836965U, 714U, 3783265725U }, + { 556206671U, 994U, 799626459U }, + { 684907218U, 799U, 367789491U }, + { 2121230701U, 931U, 2115467001U }, + { 1668516451U, 644U, 3620590685U }, + { 768046066U, 883U, 2034077390U }, + { 1989159136U, 833U, 1195767305U }, + { 536585145U, 996U, 3577259204U }, + { 1008129373U, 642U, 1478080776U }, + { 1740775604U, 939U, 1264980372U }, + { 1967883163U, 508U, 10734624U }, + { 1923019697U, 730U, 3821419629U }, + { 442079932U, 560U, 3440032343U }, + { 1961302714U, 845U, 841962572U }, + { 2030205964U, 962U, 1325144227U }, + { 1160407529U, 507U, 240940858U }, + { 635482502U, 779U, 4200489746U }, + { 1252788931U, 699U, 867195434U }, + { 1961817131U, 719U, 668237657U }, + { 1071468216U, 983U, 917876630U }, + { 1281848367U, 932U, 1003100039U }, + { 582537119U, 780U, 1127273778U }, + { 1973672777U, 853U, 1071368872U }, + { 1896756996U, 762U, 1127851055U }, + { 847917054U, 500U, 1717499075U }, + { 1240520510U, 951U, 2849576657U }, + { 1685071682U, 567U, 1961810396U }, + { 1516232129U, 557U, 3173877U }, + { 1208118903U, 612U, 1613145022U }, + { 1817269927U, 693U, 4279122573U }, + { 1510091701U, 717U, 638191229U }, + { 365916850U, 807U, 600424314U }, + { 399324359U, 702U, 1803598116U }, + { 1318480274U, 779U, 2074237022U }, + { 697758115U, 840U, 1483639402U }, + { 1696507773U, 840U, 577415447U }, + { 2081979121U, 981U, 3041486449U }, + { 955646687U, 742U, 3846494357U }, + { 1250683506U, 749U, 836419859U }, + { 595003102U, 534U, 366794109U }, + { 47485338U, 558U, 3521120834U }, + { 619433479U, 610U, 3991783875U }, + { 704096520U, 518U, 4139493852U }, + { 1712224984U, 606U, 2393312003U }, + { 1318233152U, 922U, 3880361134U }, + { 855572992U, 761U, 1472974787U }, + { 64721421U, 703U, 683860550U }, + { 678931758U, 840U, 380616043U }, + { 692711973U, 778U, 1382361947U }, + { 677703619U, 530U, 2826914161U }, + { 92393223U, 586U, 1522128471U }, + { 1222592920U, 743U, 3466726667U }, + { 358288986U, 695U, 1091956998U }, + { 1935056945U, 958U, 514864477U }, + { 735675993U, 990U, 1294239989U }, + { 1560089402U, 897U, 2238551287U }, + { 70616361U, 829U, 22483098U }, + { 368234700U, 731U, 2913875084U }, + { 20221190U, 879U, 1564152970U }, + { 539444654U, 682U, 1835141259U }, + { 1314987297U, 840U, 1801114136U }, + { 2019295544U, 645U, 3286438930U }, + { 469023838U, 716U, 1637918202U }, + { 1843754496U, 653U, 2562092152U }, + { 400672036U, 809U, 4264212785U }, + { 404722249U, 965U, 2704116999U }, + { 600702209U, 758U, 584979986U }, + { 519953954U, 667U, 2574436237U }, + { 1658071126U, 694U, 2214569490U }, + { 420480037U, 749U, 3430010866U }, + { 690103647U, 969U, 3700758083U }, + { 1029424799U, 937U, 3787746841U }, + { 2012608669U, 506U, 3362628973U }, + { 1535432887U, 998U, 42610943U }, + { 1330635533U, 857U, 3040806504U }, + { 1223800550U, 539U, 3954229517U }, + { 1322411537U, 680U, 3223250324U }, + { 1877847898U, 945U, 2915147143U }, + { 1646356099U, 874U, 965988280U }, + { 805687536U, 744U, 4032277920U }, + { 1948093210U, 633U, 1346597684U }, + { 392609744U, 783U, 1636083295U }, + { 690241304U, 770U, 1201031298U }, + { 1360302965U, 696U, 1665394461U }, + { 1220090946U, 780U, 1316922812U }, + { 447092251U, 500U, 3438743375U }, + { 1613868791U, 592U, 828546883U }, + { 523430951U, 548U, 2552392304U }, + { 726692899U, 810U, 1656872867U }, + { 1364340021U, 836U, 3710513486U }, + { 1986257729U, 931U, 935013962U }, + { 407983964U, 921U, 728767059U }, +}; + +static void prandom_state_test_seed(struct rnd_state *state, u32 seed) +{ +#define LCG(x) ((x) * 69069U) /* super-duper LCG */ + state->s1 = __seed(LCG(seed), 2U); + state->s2 = __seed(LCG(state->s1), 8U); + state->s3 = __seed(LCG(state->s2), 16U); + state->s4 = __seed(LCG(state->s3), 128U); +} + +static void test_prandom_seed_boundary(struct kunit *test) +{ + int i; + struct rnd_state state; + + for (i = 0; i < ARRAY_SIZE(test1); i++) { + prandom_state_test_seed(&state, test1[i].seed); + prandom_warmup(&state); + KUNIT_EXPECT_EQ(test, test1[i].result, prandom_u32_state(&state)); + } +} + +static void test_prandom_taus113(struct kunit *test) +{ + int i, j; + struct rnd_state state; + + for (i = 0; i < ARRAY_SIZE(test2); i++) { + prandom_state_test_seed(&state, test2[i].seed); + prandom_warmup(&state); + + for (j = 0; j < test2[i].iteration - 1; j++) + prandom_u32_state(&state); + + KUNIT_EXPECT_EQ(test, test2[i].result, prandom_u32_state(&state)); + } +} + +static struct kunit_case prandom_test_cases[] = { + KUNIT_CASE(test_prandom_seed_boundary), + KUNIT_CASE(test_prandom_taus113), + {} +}; + +static struct kunit_suite prandom_test_suite = { + .name = "prandom", + .test_cases = prandom_test_cases, +}; + +kunit_test_suite(prandom_test_suite); + +MODULE_DESCRIPTION("KUnit test for prandom"); +MODULE_LICENSE("GPL"); +MODULE_IMPORT_NS("EXPORTED_FOR_KUNIT_TESTING"); diff --git a/lib/ucs2_string.c b/lib/ucs2_string.c index dfb4f2358cab..1f7dd4eb640a 100644 --- a/lib/ucs2_string.c +++ b/lib/ucs2_string.c @@ -6,18 +6,18 @@ unsigned long ucs2_strnlen(const ucs2_char_t *s, size_t maxlength) { - unsigned long length = 0; + unsigned long length = 0; - while (*s++ != 0 && length < maxlength) - length++; - return length; + while (length < maxlength && *s++ != 0) + length++; + return length; } EXPORT_SYMBOL(ucs2_strnlen); unsigned long ucs2_strlen(const ucs2_char_t *s) { - return ucs2_strnlen(s, ~0UL); + return ucs2_strnlen(s, ~0UL); } EXPORT_SYMBOL(ucs2_strlen); @@ -28,7 +28,7 @@ EXPORT_SYMBOL(ucs2_strlen); unsigned long ucs2_strsize(const ucs2_char_t *data, unsigned long maxlength) { - return ucs2_strnlen(data, maxlength/sizeof(ucs2_char_t)) * sizeof(ucs2_char_t); + return ucs2_strnlen(data, maxlength/sizeof(ucs2_char_t)) * sizeof(ucs2_char_t); } EXPORT_SYMBOL(ucs2_strsize); @@ -87,19 +87,19 @@ EXPORT_SYMBOL(ucs2_strscpy); int ucs2_strncmp(const ucs2_char_t *a, const ucs2_char_t *b, size_t len) { - while (1) { - if (len == 0) - return 0; - if (*a < *b) - return -1; - if (*a > *b) - return 1; - if (*a == 0) /* implies *b == 0 */ - return 0; - a++; - b++; - len--; - } + while (1) { + if (len == 0) + return 0; + if (*a < *b) + return -1; + if (*a > *b) + return 1; + if (*a == 0) /* implies *b == 0 */ + return 0; + a++; + b++; + len--; + } } EXPORT_SYMBOL(ucs2_strncmp); diff --git a/lib/xz/xz_dec_bcj.c b/lib/xz/xz_dec_bcj.c index cc49a300a5b2..88922323f96e 100644 --- a/lib/xz/xz_dec_bcj.c +++ b/lib/xz/xz_dec_bcj.c @@ -466,7 +466,7 @@ static void bcj_flush(struct xz_dec_bcj *s, struct xz_buf *b) { size_t copy_size; - copy_size = min_t(size_t, s->temp.filtered, b->out_size - b->out_pos); + copy_size = min(s->temp.filtered, b->out_size - b->out_pos); memcpy(b->out + b->out_pos, s->temp.buf, copy_size); b->out_pos += copy_size; diff --git a/lib/xz/xz_dec_lzma2.c b/lib/xz/xz_dec_lzma2.c index 4b783ac94e71..46c2df6ad6f5 100644 --- a/lib/xz/xz_dec_lzma2.c +++ b/lib/xz/xz_dec_lzma2.c @@ -135,14 +135,16 @@ struct lzma_dec { uint32_t rep2; uint32_t rep3; - /* Types of the most recently seen LZMA symbols */ - enum lzma_state state; - /* * Length of a match. This is updated so that dict_repeat can - * be called again to finish repeating the whole match. + * be called again to finish repeating the whole match. This is + * size_t because a pointer to this is passed to dict_repeat, + * and there it's nicer to have size_t instead of uint32_t. */ - uint32_t len; + size_t len; + + /* Types of the most recently seen LZMA symbols */ + enum lzma_state state; /* * LZMA properties or related bit masks (number of literal @@ -228,13 +230,13 @@ struct lzma2_dec { enum lzma2_seq next_sequence; /* Uncompressed size of LZMA chunk (2 MiB at maximum) */ - uint32_t uncompressed; + size_t uncompressed; /* * Compressed size of LZMA chunk or compressed/uncompressed * size of uncompressed chunk (64 KiB at maximum) */ - uint32_t compressed; + size_t compressed; /* * True if dictionary reset is needed. This is false before @@ -269,11 +271,11 @@ struct xz_dec_lzma2 { struct lzma_dec lzma; /* - * Temporary buffer which holds small number of input bytes between + * Temporary buffer which holds a small number of input bytes between * decoder calls. See lzma2_lzma() for details. */ struct { - uint32_t size; + size_t size; uint8_t buf[3 * LZMA_IN_REQUIRED]; } temp; }; @@ -320,7 +322,7 @@ static inline bool dict_has_space(const struct dictionary *dict) * still empty. This special case is needed for single-call decoding to * avoid writing a '\0' to the end of the destination buffer. */ -static inline uint32_t dict_get(const struct dictionary *dict, uint32_t dist) +static inline uint32_t dict_get(const struct dictionary *dict, size_t dist) { size_t offset = dict->pos - dist - 1; @@ -346,15 +348,15 @@ static inline void dict_put(struct dictionary *dict, uint8_t byte) * invalid, false is returned. On success, true is returned and *len is * updated to indicate how many bytes were left to be repeated. */ -static bool dict_repeat(struct dictionary *dict, uint32_t *len, uint32_t dist) +static bool dict_repeat(struct dictionary *dict, size_t *len, size_t dist) { size_t back; - uint32_t left; + size_t left; if (dist >= dict->full || dist >= dict->size) return false; - left = min_t(size_t, dict->limit - dict->pos, *len); + left = min(dict->limit - dict->pos, *len); *len -= left; back = dict->pos - dist - 1; @@ -375,7 +377,7 @@ static bool dict_repeat(struct dictionary *dict, uint32_t *len, uint32_t dist) /* Copy uncompressed data as is from input to dictionary and output buffers. */ static void dict_uncompressed(struct dictionary *dict, struct xz_buf *b, - uint32_t *left) + size_t *left) { size_t copy_size; @@ -433,7 +435,7 @@ static void dict_uncompressed(struct dictionary *dict, struct xz_buf *b, * enough space in b->out. This is guaranteed because caller uses dict_limit() * before decoding data into the dictionary. */ -static uint32_t dict_flush(struct dictionary *dict, struct xz_buf *b) +static size_t dict_flush(struct dictionary *dict, struct xz_buf *b) { size_t copy_size = dict->pos - dict->start; @@ -755,8 +757,8 @@ static bool lzma_main(struct xz_dec_lzma2 *s) uint32_t pos_state; /* - * If the dictionary was reached during the previous call, try to - * finish the possibly pending repeat in the dictionary. + * If the dictionary write limit was reached during the previous call, + * try to finish the possibly pending repeat in the dictionary. */ if (dict_has_space(&s->dict) && s->lzma.len > 0) dict_repeat(&s->dict, &s->lzma.len, s->lzma.rep0); @@ -878,7 +880,7 @@ static bool lzma_props(struct xz_dec_lzma2 *s, uint8_t props) static bool lzma2_lzma(struct xz_dec_lzma2 *s, struct xz_buf *b) { size_t in_avail; - uint32_t tmp; + size_t tmp; in_avail = b->in_size - b->in_pos; if (s->temp.size > 0 || s->lzma2.compressed == 0) { @@ -976,7 +978,7 @@ enum xz_ret xz_dec_lzma2_run(struct xz_dec_lzma2 *s, struct xz_buf *b) * an uncompressed chunk * 0x02 Uncompressed chunk (no dictionary reset) * - * Highest three bits (s->control & 0xE0): + * Highest three bits (tmp & 0xE0): * 0xE0 Dictionary reset, new properties and state * reset, followed by LZMA compressed chunk * 0xC0 New properties and state reset, followed @@ -988,7 +990,7 @@ enum xz_ret xz_dec_lzma2_run(struct xz_dec_lzma2 *s, struct xz_buf *b) * 0x80 LZMA chunk (no dictionary or state reset) * * For LZMA compressed chunks, the lowest five bits - * (s->control & 1F) are the highest bits of the + * (tmp & 1F) are the highest bits of the * uncompressed size (bits 16-20). * * A new LZMA2 stream must begin with a dictionary @@ -1046,25 +1048,23 @@ enum xz_ret xz_dec_lzma2_run(struct xz_dec_lzma2 *s, struct xz_buf *b) case SEQ_UNCOMPRESSED_1: s->lzma2.uncompressed - += (uint32_t)b->in[b->in_pos++] << 8; + += (size_t)b->in[b->in_pos++] << 8; s->lzma2.sequence = SEQ_UNCOMPRESSED_2; break; case SEQ_UNCOMPRESSED_2: s->lzma2.uncompressed - += (uint32_t)b->in[b->in_pos++] + 1; + += (size_t)b->in[b->in_pos++] + 1; s->lzma2.sequence = SEQ_COMPRESSED_0; break; case SEQ_COMPRESSED_0: - s->lzma2.compressed - = (uint32_t)b->in[b->in_pos++] << 8; + s->lzma2.compressed = (size_t)b->in[b->in_pos++] << 8; s->lzma2.sequence = SEQ_COMPRESSED_1; break; case SEQ_COMPRESSED_1: - s->lzma2.compressed - += (uint32_t)b->in[b->in_pos++] + 1; + s->lzma2.compressed += (size_t)b->in[b->in_pos++] + 1; s->lzma2.sequence = s->lzma2.next_sequence; break; @@ -1091,16 +1091,15 @@ enum xz_ret xz_dec_lzma2_run(struct xz_dec_lzma2 *s, struct xz_buf *b) case SEQ_LZMA_RUN: /* * Set dictionary limit to indicate how much we want - * to be encoded at maximum. Decode new data into the + * to be decoded at maximum. Decode new data into the * dictionary. Flush the new data from dictionary to * b->out. Check if we finished decoding this chunk. * In case the dictionary got full but we didn't fill * the output buffer yet, we may run this loop * multiple times without changing s->lzma2.sequence. */ - dict_limit(&s->dict, min_t(size_t, - b->out_size - b->out_pos, - s->lzma2.uncompressed)); + dict_limit(&s->dict, min(b->out_size - b->out_pos, + s->lzma2.uncompressed)); if (!lzma2_lzma(s, b)) return XZ_DATA_ERROR; @@ -1260,8 +1259,8 @@ enum xz_ret xz_dec_microlzma_run(struct xz_dec_microlzma *s_ptr, s->dict.end = b->out_size - b->out_pos; while (true) { - dict_limit(&s->dict, min_t(size_t, b->out_size - b->out_pos, - s->lzma2.uncompressed)); + dict_limit(&s->dict, min(b->out_size - b->out_pos, + s->lzma2.uncompressed)); if (!lzma2_lzma(s, b)) return XZ_DATA_ERROR; diff --git a/lib/xz/xz_dec_stream.c b/lib/xz/xz_dec_stream.c index 59bfd54ffee7..0bed6daefac2 100644 --- a/lib/xz/xz_dec_stream.c +++ b/lib/xz/xz_dec_stream.c @@ -155,8 +155,8 @@ static const uint8_t check_sizes[16] = { */ static bool fill_temp(struct xz_dec *s, struct xz_buf *b) { - size_t copy_size = min_t(size_t, - b->in_size - b->in_pos, s->temp.size - s->temp.pos); + size_t copy_size = min(b->in_size - b->in_pos, + s->temp.size - s->temp.pos); memcpy(s->temp.buf + s->temp.pos, b->in + b->in_pos, copy_size); b->in_pos += copy_size; diff --git a/lib/xz/xz_lzma2.h b/lib/xz/xz_lzma2.h index d2632b7dfb9c..a612ce4fd450 100644 --- a/lib/xz/xz_lzma2.h +++ b/lib/xz/xz_lzma2.h @@ -143,7 +143,7 @@ static inline bool lzma_state_is_literal(enum lzma_state state) * Get the index of the appropriate probability array for decoding * the distance slot. */ -static inline uint32_t lzma_get_dist_state(uint32_t len) +static inline size_t lzma_get_dist_state(size_t len) { return len < DIST_STATES + MATCH_LEN_MIN ? len - MATCH_LEN_MIN : DIST_STATES - 1; diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 7961d9584bc5..8a7787d228a6 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -3782,6 +3782,12 @@ sub process { my $vp_file = $dt_path . "vendor-prefixes.yaml"; foreach my $compat (@compats) { + # Skip ID-based PCI and USB compatible patterns. + # DT validation will check them properly. + next if $compat =~ /^pciclass,/; + next if $compat =~ /^pci[a-f0-9]{2,4},/; + next if $compat =~ /^usb(if)?[a-f0-9]{1,4},/; + my $compat2 = $compat; $compat2 =~ s/\,[a-zA-Z0-9]*\-/\,<\.\*>\-/; my $compat3 = $compat; @@ -4149,6 +4155,7 @@ sub process { $line =~ /^\+[a-z_]*init/ || $line =~ /^\+\s*(?:static\s+)?[A-Z_]*ATTR/ || $line =~ /^\+\s*DECLARE/ || + $line =~ /^\+\s*NOKPROBE_SYMBOL/ || $line =~ /^\+\s*builtin_[\w_]*driver/ || $line =~ /^\+\s*__setup/)) { if (CHK("LINE_SPACING", diff --git a/tools/accounting/delaytop.c b/tools/accounting/delaytop.c index 72cc500b44b1..0a863dfeae8a 100644 --- a/tools/accounting/delaytop.c +++ b/tools/accounting/delaytop.c @@ -878,7 +878,7 @@ static void display_results(int psi_ret) suc &= BOOL_FPRINT(out, "\033[H\033[J"); /* PSI output (one-line, no cat style) */ - suc &= BOOL_FPRINT(out, "System Pressure Information: (avg10/avg60vg300/total)\n"); + suc &= BOOL_FPRINT(out, "System Pressure Information: (avg10/avg60/avg300/total)\n"); if (psi_ret) { suc &= BOOL_FPRINT(out, " PSI not found: check if psi=1 enabled in cmdline\n"); } else { diff --git a/tools/accounting/getdelays.c b/tools/accounting/getdelays.c index caa5fe9dd573..52930629ec3f 100644 --- a/tools/accounting/getdelays.c +++ b/tools/accounting/getdelays.c @@ -302,7 +302,7 @@ static const char *format_timespec(struct __kernel_timespec *ts) average_ms((double)(t)->cpu_delay_total, (t)->cpu_count)); \ } \ } while (0) -#define PRINT_FILED_DELAY(name, version, t, count, total, max, min) \ +#define PRINT_FIELD_DELAY(name, version, t, count, total, max, min) \ do { \ if (version >= 16) { \ printf("%-10s%15s%15s%15s%15s%15s\n", \ @@ -324,7 +324,7 @@ static const char *format_timespec(struct __kernel_timespec *ts) } \ } while (0) -#define PRINT_FILED_DELAY_WITH_TS(name, version, t, count, total, max, min, max_ts) \ +#define PRINT_FIELD_DELAY_WITH_TS(name, version, t, count, total, max, min, max_ts) \ do { \ if (version >= 17) { \ printf("%-10s%15s%15s%15s%15s%15s%25s\n", \ @@ -365,71 +365,71 @@ static void print_delayacct(struct taskstats *t) /* Use new macro with timestamp support for version >= 17 */ if (t->version >= 17) { - PRINT_FILED_DELAY_WITH_TS("IO", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("IO", t->version, t, blkio_count, blkio_delay_total, blkio_delay_max, blkio_delay_min, blkio_delay_max_ts); - PRINT_FILED_DELAY_WITH_TS("SWAP", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("SWAP", t->version, t, swapin_count, swapin_delay_total, swapin_delay_max, swapin_delay_min, swapin_delay_max_ts); - PRINT_FILED_DELAY_WITH_TS("RECLAIM", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("RECLAIM", t->version, t, freepages_count, freepages_delay_total, freepages_delay_max, freepages_delay_min, freepages_delay_max_ts); - PRINT_FILED_DELAY_WITH_TS("THRASHING", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("THRASHING", t->version, t, thrashing_count, thrashing_delay_total, thrashing_delay_max, thrashing_delay_min, thrashing_delay_max_ts); if (t->version >= 11) { - PRINT_FILED_DELAY_WITH_TS("COMPACT", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("COMPACT", t->version, t, compact_count, compact_delay_total, compact_delay_max, compact_delay_min, compact_delay_max_ts); } if (t->version >= 13) { - PRINT_FILED_DELAY_WITH_TS("WPCOPY", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("WPCOPY", t->version, t, wpcopy_count, wpcopy_delay_total, wpcopy_delay_max, wpcopy_delay_min, wpcopy_delay_max_ts); } if (t->version >= 14) { - PRINT_FILED_DELAY_WITH_TS("IRQ", t->version, t, + PRINT_FIELD_DELAY_WITH_TS("IRQ", t->version, t, irq_count, irq_delay_total, irq_delay_max, irq_delay_min, irq_delay_max_ts); } } else { /* Use original macro for older versions */ - PRINT_FILED_DELAY("IO", t->version, t, + PRINT_FIELD_DELAY("IO", t->version, t, blkio_count, blkio_delay_total, blkio_delay_max, blkio_delay_min); - PRINT_FILED_DELAY("SWAP", t->version, t, + PRINT_FIELD_DELAY("SWAP", t->version, t, swapin_count, swapin_delay_total, swapin_delay_max, swapin_delay_min); - PRINT_FILED_DELAY("RECLAIM", t->version, t, + PRINT_FIELD_DELAY("RECLAIM", t->version, t, freepages_count, freepages_delay_total, freepages_delay_max, freepages_delay_min); - PRINT_FILED_DELAY("THRASHING", t->version, t, + PRINT_FIELD_DELAY("THRASHING", t->version, t, thrashing_count, thrashing_delay_total, thrashing_delay_max, thrashing_delay_min); if (t->version >= 11) { - PRINT_FILED_DELAY("COMPACT", t->version, t, + PRINT_FIELD_DELAY("COMPACT", t->version, t, compact_count, compact_delay_total, compact_delay_max, compact_delay_min); } if (t->version >= 13) { - PRINT_FILED_DELAY("WPCOPY", t->version, t, + PRINT_FIELD_DELAY("WPCOPY", t->version, t, wpcopy_count, wpcopy_delay_total, wpcopy_delay_max, wpcopy_delay_min); } if (t->version >= 14) { - PRINT_FILED_DELAY("IRQ", t->version, t, + PRINT_FIELD_DELAY("IRQ", t->version, t, irq_count, irq_delay_total, irq_delay_max, irq_delay_min); } diff --git a/tools/include/linux/compiler.h b/tools/include/linux/compiler.h index f40bd2b04c29..f2f54b038168 100644 --- a/tools/include/linux/compiler.h +++ b/tools/include/linux/compiler.h @@ -119,7 +119,7 @@ #define __read_mostly #ifndef __attribute_const__ -# define __attribute_const__ +# define __attribute_const__ __attribute__ ((__const__)) #endif #ifndef __maybe_unused diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index 35d3d254941c..22b3b500d33a 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -836,8 +836,7 @@ int main(int argc, char **argv) } fin = fopen(argv[optind], "r"); - fout = fopen(argv[optind + 1], "w"); - if (!fin || !fout) { + if (!fin) { usage(); perror("open: "); exit(1); @@ -874,6 +873,13 @@ int main(int argc, char **argv) goto out_free; } + fout = fopen(argv[optind + 1], "w"); + if (!fout) { + usage(); + perror("open: "); + exit(1); + } + printf("loaded %d\n", list_size); printf("sorting ....\n"); diff --git a/tools/testing/selftests/acct/.gitignore b/tools/testing/selftests/acct/.gitignore index 9e9c61c5bfd6..fe0896f54e15 100644 --- a/tools/testing/selftests/acct/.gitignore +++ b/tools/testing/selftests/acct/.gitignore @@ -1,4 +1,5 @@ acct_syscall taskstats_fill_stats_tgid +cgroupstats config process_log diff --git a/tools/testing/selftests/acct/Makefile b/tools/testing/selftests/acct/Makefile index 083cab5ddb72..93a11a28a636 100644 --- a/tools/testing/selftests/acct/Makefile +++ b/tools/testing/selftests/acct/Makefile @@ -1,8 +1,19 @@ # SPDX-License-Identifier: GPL-2.0 TEST_GEN_PROGS := acct_syscall TEST_GEN_PROGS += taskstats_fill_stats_tgid +TEST_GEN_PROGS += cgroupstats + +NETLINK_HELPER_PROGS := cgroupstats taskstats_fill_stats_tgid CFLAGS += -Wall LDLIBS += -lpthread include ../lib.mk + +$(NETLINK_HELPER_PROGS): %: %.c netlink_helper.c netlink_helper.h + $(call msg,CC,,$@) + $(Q)$(LINK.c) $< netlink_helper.c $(LDLIBS) -o $@ + +$(addprefix $(OUTPUT)/,$(NETLINK_HELPER_PROGS)): $(OUTPUT)/%: %.c netlink_helper.c netlink_helper.h + $(call msg,CC,,$@) + $(Q)$(LINK.c) $< netlink_helper.c $(LDLIBS) -o $@ diff --git a/tools/testing/selftests/acct/cgroupstats.c b/tools/testing/selftests/acct/cgroupstats.c new file mode 100644 index 000000000000..0b421a4ca72b --- /dev/null +++ b/tools/testing/selftests/acct/cgroupstats.c @@ -0,0 +1,231 @@ +// SPDX-License-Identifier: GPL-2.0 +#define _GNU_SOURCE + +#include <errno.h> +#include <fcntl.h> +#include <linux/cgroupstats.h> +#include <linux/genetlink.h> +#include <linux/netlink.h> +#include <stdbool.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/mount.h> +#include <sys/socket.h> +#include <sys/types.h> +#include <unistd.h> + +#include "netlink_helper.h" +#include "kselftest.h" + +static int send_cgroupstats_cmd(int fd, int family_id, uint32_t cgroup_fd, + int flags) +{ + struct { + struct nlmsghdr nlh; + struct genlmsghdr genl; + char buf[256]; + } req = { 0 }; + struct nlattr *na; + + req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN); + req.nlh.nlmsg_type = family_id; + req.nlh.nlmsg_flags = NLM_F_REQUEST | flags; + req.nlh.nlmsg_seq = 2; + req.nlh.nlmsg_pid = getpid(); + + req.genl.cmd = CGROUPSTATS_CMD_GET; + req.genl.version = 1; + + na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len)); + na->nla_type = CGROUPSTATS_CMD_ATTR_FD; + na->nla_len = NLA_HDRLEN + sizeof(cgroup_fd); + memcpy(nla_data(na), &cgroup_fd, sizeof(cgroup_fd)); + req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len); + + return send_request(fd, &req, req.nlh.nlmsg_len); +} + +/* + * Receive and decode a cgroupstats response. + * + * Returns: + * 0 — success, stats filled from CGROUPSTATS_CMD_NEW reply + * <0 — NLMSG_ERROR errno (e.g. -EBADF, -EINVAL) + */ +static int recv_cgroupstats_response(int fd, struct cgroupstats *stats) +{ + char resp[8192]; + struct nlmsghdr *nlh; + struct genlmsghdr *genl; + struct nlattr *na; + int len; + int rem; + + memset(stats, 0, sizeof(*stats)); + + len = recv(fd, resp, sizeof(resp), 0); + if (len < 0) + return -errno; + + for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len); + nlh = NLMSG_NEXT(nlh, len)) { + if (nlh->nlmsg_type == NLMSG_ERROR) { + struct nlmsgerr *err = NLMSG_DATA(nlh); + + return err->error; + } + + genl = (struct genlmsghdr *)NLMSG_DATA(nlh); + if (genl->cmd != CGROUPSTATS_CMD_NEW) + continue; + + rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; + na = (struct nlattr *)((char *)genl + GENL_HDRLEN); + while (nla_ok(na, rem)) { + if (na->nla_type == CGROUPSTATS_TYPE_CGROUP_STATS) { + memcpy(stats, nla_data(na), sizeof(*stats)); + return 0; + } + na = nla_next(na, &rem); + } + } + + return -EIO; +} + +/* mkdtemp() modifies the template in place, so this cannot be const. */ +static char cg_mountpoint[32]; +static bool cg_mounted; + +static int setup_cgroup_v1(void) +{ + strcpy(cg_mountpoint, "/tmp/cgstats_test_XXXXXX"); + + if (!mkdtemp(cg_mountpoint)) + return -errno; + + if (mount("cgstats_test", cg_mountpoint, "cgroup", 0, + "none,name=cgstats_test") < 0) { + int ret = -errno; + + rmdir(cg_mountpoint); + return ret; + } + + cg_mounted = true; + return 0; +} + +static void cleanup_cgroup_v1(void) +{ + if (!cg_mounted) + return; + umount2(cg_mountpoint, MNT_DETACH); + rmdir(cg_mountpoint); + cg_mounted = false; +} + +int main(void) +{ + struct cgroupstats stats; + uint64_t total_tasks; + int family_id; + int nl_fd; + int cg_fd; + int ret; + + ksft_print_header(); + + nl_fd = netlink_open(); + if (nl_fd < 0) + ksft_exit_skip("failed to open generic netlink socket: %s\n", + strerror(-nl_fd)); + + family_id = get_family_id(nl_fd, TASKSTATS_GENL_NAME); + if (family_id < 0) + ksft_exit_skip("taskstats generic netlink family unavailable: %s\n", + strerror(-family_id)); + + ksft_set_plan(3); + + /* + * Test 1: mount a private cgroup v1 hierarchy, query it, and + * verify the response contains sane task counts. If the test + * environment cannot create a private cgroup v1 mount, skip this + * case and continue with the unprivileged regression checks below. + */ + ret = setup_cgroup_v1(); + if (ret) { + ksft_test_result_skip("cgroupstats query: cannot mount cgroup v1: %s\n", + strerror(-ret)); + } else { + cg_fd = open(cg_mountpoint, O_RDONLY | O_DIRECTORY); + if (cg_fd < 0) { + ksft_test_result_fail("cgroupstats query: open mountpoint: %s\n", + strerror(errno)); + } else { + ret = send_cgroupstats_cmd(nl_fd, family_id, + (uint32_t)cg_fd, 0); + if (ret) { + ksft_test_result_fail("cgroupstats query: send: %s\n", + strerror(-ret)); + } else { + ret = recv_cgroupstats_response(nl_fd, &stats); + if (ret < 0) { + ksft_test_result_fail("cgroupstats query: %s\n", + strerror(-ret)); + } else { + total_tasks = (uint64_t)stats.nr_sleeping + + (uint64_t)stats.nr_running + + (uint64_t)stats.nr_stopped + + (uint64_t)stats.nr_uninterruptible + + (uint64_t)stats.nr_io_wait; + + ksft_print_msg("cgroupstats query: total_tasks=%llu\n", + (unsigned long long)total_tasks); + + ksft_test_result(total_tasks > 0, + "cgroupstats query returns valid stats\n"); + } + } + close(cg_fd); + } + } + cleanup_cgroup_v1(); + + /* + * Test 2: invalid fd without NLM_F_ACK. The kernel should + * return -EBADF via NLMSG_ERROR regardless of whether the + * client requested an explicit ACK. + */ + ret = send_cgroupstats_cmd(nl_fd, family_id, 0xFFFFFFFF, 0); + if (ret) + ksft_exit_fail_msg("send test 2 failed: %s\n", strerror(-ret)); + + ret = recv_cgroupstats_response(nl_fd, &stats); + ksft_print_msg("bad fd (no ACK): response=%d (%s)\n", + ret, ret < 0 ? strerror(-ret) : "unexpected success"); + ksft_test_result(ret == -EBADF, + "cgroupstats rejects bad fd without NLM_F_ACK\n"); + + /* + * Test 3: invalid fd with NLM_F_ACK. Same expectation as + * test 2, but exercised through a different netlink flag + * path in the kernel's ack/error handling. + */ + ret = send_cgroupstats_cmd(nl_fd, family_id, 0xFFFFFFFF, NLM_F_ACK); + if (ret) + ksft_exit_fail_msg("send test 3 failed: %s\n", strerror(-ret)); + + ret = recv_cgroupstats_response(nl_fd, &stats); + ksft_print_msg("bad fd (with ACK): response=%d (%s)\n", + ret, ret < 0 ? strerror(-ret) : "unexpected success"); + ksft_test_result(ret == -EBADF, + "cgroupstats rejects bad fd with NLM_F_ACK\n"); + + close(nl_fd); + ksft_finished(); + return ksft_get_fail_cnt() ? KSFT_FAIL : KSFT_PASS; +} diff --git a/tools/testing/selftests/acct/netlink_helper.c b/tools/testing/selftests/acct/netlink_helper.c new file mode 100644 index 000000000000..3ed834f0e770 --- /dev/null +++ b/tools/testing/selftests/acct/netlink_helper.c @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <errno.h> +#include <stdint.h> +#include <string.h> +#include <sys/socket.h> +#include <sys/time.h> +#include <unistd.h> +#include <linux/genetlink.h> + +#include "netlink_helper.h" + +int netlink_open(void) +{ + struct timeval tv = { .tv_sec = ACCT_RCV_TIMEOUT_SEC }; + struct sockaddr_nl addr = { + .nl_family = AF_NETLINK, + .nl_pid = getpid(), + }; + int fd; + + fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC); + if (fd < 0) + return -errno; + + if (setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)) < 0) { + int err = -errno; + + close(fd); + return err; + } + + if (bind(fd, (struct sockaddr *)&addr, sizeof(addr)) < 0) { + int err = -errno; + + close(fd); + return err; + } + + return fd; +} + +int send_request(int fd, void *buf, size_t len) +{ + struct sockaddr_nl addr = { + .nl_family = AF_NETLINK, + }; + + if (sendto(fd, buf, len, 0, (struct sockaddr *)&addr, sizeof(addr)) < 0) + return -errno; + + return 0; +} + +/* + * Resolve the generic netlink family ID for @name. + * Returns the family ID (>= 0) on success, negative errno on failure. + */ +int get_family_id(int fd, const char *name) +{ + struct { + struct nlmsghdr nlh; + struct genlmsghdr genl; + char buf[256]; + } req = { 0 }; + char resp[8192]; + struct nlmsghdr *nlh; + struct genlmsghdr *genl; + struct nlattr *na; + int len; + int rem; + int ret; + + req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN); + req.nlh.nlmsg_type = GENL_ID_CTRL; + req.nlh.nlmsg_flags = NLM_F_REQUEST; + req.nlh.nlmsg_seq = 1; + req.nlh.nlmsg_pid = getpid(); + + req.genl.cmd = CTRL_CMD_GETFAMILY; + req.genl.version = 1; + + na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len)); + na->nla_type = CTRL_ATTR_FAMILY_NAME; + na->nla_len = NLA_HDRLEN + strlen(name) + 1; + memcpy(nla_data(na), name, strlen(name) + 1); + req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len); + + ret = send_request(fd, &req, req.nlh.nlmsg_len); + if (ret) + return ret; + + len = recv(fd, resp, sizeof(resp), 0); + if (len < 0) + return -errno; + + for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len); + nlh = NLMSG_NEXT(nlh, len)) { + if (nlh->nlmsg_type == NLMSG_ERROR) { + struct nlmsgerr *err = NLMSG_DATA(nlh); + + return err->error ? err->error : -ENOENT; + } + + genl = (struct genlmsghdr *)NLMSG_DATA(nlh); + rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; + na = (struct nlattr *)((char *)genl + GENL_HDRLEN); + while (nla_ok(na, rem)) { + if (na->nla_type == CTRL_ATTR_FAMILY_ID) + return *(uint16_t *)nla_data(na); + na = nla_next(na, &rem); + } + } + + return -ENOENT; +} diff --git a/tools/testing/selftests/acct/netlink_helper.h b/tools/testing/selftests/acct/netlink_helper.h new file mode 100644 index 000000000000..0320729c4c06 --- /dev/null +++ b/tools/testing/selftests/acct/netlink_helper.h @@ -0,0 +1,44 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Shared generic netlink helpers for the acct selftests. + */ +#ifndef ACSELFTESTS_ACCT_NETLINK_HELPER_H +#define ACSELFTESTS_ACCT_NETLINK_HELPER_H + +#include <stdbool.h> +#include <linux/netlink.h> + +#ifndef NLA_ALIGNTO +#define NLA_ALIGNTO 4 +#define NLA_ALIGN(len) (((len) + NLA_ALIGNTO - 1) & ~(NLA_ALIGNTO - 1)) +#define NLA_HDRLEN ((int)NLA_ALIGN(sizeof(struct nlattr))) +#endif + +/* Fail an individual test case instead of hanging the whole binary. */ +#define ACCT_RCV_TIMEOUT_SEC 2 + +static inline void *nla_data(const struct nlattr *na) +{ + return (void *)((char *)na + NLA_HDRLEN); +} + +static inline bool nla_ok(const struct nlattr *na, int remaining) +{ + return remaining >= (int)sizeof(*na) && + na->nla_len >= sizeof(*na) && + na->nla_len <= remaining; +} + +static inline struct nlattr *nla_next(const struct nlattr *na, int *remaining) +{ + int aligned_len = NLA_ALIGN(na->nla_len); + + *remaining -= aligned_len; + return (struct nlattr *)((char *)na + aligned_len); +} + +int netlink_open(void); +int send_request(int fd, void *buf, size_t len); +int get_family_id(int fd, const char *name); + +#endif /* ACSELFTESTS_ACCT_NETLINK_HELPER_H */ diff --git a/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c b/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c index d6cab4ae26f2..9a4c1554dee3 100644 --- a/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c +++ b/tools/testing/selftests/acct/taskstats_fill_stats_tgid.c @@ -16,14 +16,9 @@ #include <time.h> #include <unistd.h> +#include "netlink_helper.h" #include "kselftest.h" -#ifndef NLA_ALIGN -#define NLA_ALIGNTO 4 -#define NLA_ALIGN(len) (((len) + NLA_ALIGNTO - 1) & ~(NLA_ALIGNTO - 1)) -#define NLA_HDRLEN ((int)NLA_ALIGN(sizeof(struct nlattr))) -#endif - #define BUSY_NS (200ULL * 1000 * 1000) struct worker_ctx { @@ -35,26 +30,6 @@ struct worker_ctx { static unsigned long busy_sink; -static void *taskstats_nla_data(const struct nlattr *na) -{ - return (void *)((char *)na + NLA_HDRLEN); -} - -static bool taskstats_nla_ok(const struct nlattr *na, int remaining) -{ - return remaining >= (int)sizeof(*na) && - na->nla_len >= sizeof(*na) && - na->nla_len <= remaining; -} - -static struct nlattr *taskstats_nla_next(const struct nlattr *na, int *remaining) -{ - int aligned_len = NLA_ALIGN(na->nla_len); - - *remaining -= aligned_len; - return (struct nlattr *)((char *)na + aligned_len); -} - static uint64_t timespec_diff_ns(const struct timespec *start, const struct timespec *end) { @@ -84,99 +59,6 @@ static void burn_cpu_for_ns(uint64_t runtime_ns) busy_sink = acc; } -static int netlink_open(void) -{ - struct sockaddr_nl addr = { - .nl_family = AF_NETLINK, - .nl_pid = getpid(), - }; - int fd; - - fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC); - if (fd < 0) - return -errno; - - if (bind(fd, (struct sockaddr *)&addr, sizeof(addr)) < 0) { - int err = -errno; - - close(fd); - return err; - } - - return fd; -} - -static int send_request(int fd, void *buf, size_t len) -{ - struct sockaddr_nl addr = { - .nl_family = AF_NETLINK, - }; - - if (sendto(fd, buf, len, 0, (struct sockaddr *)&addr, sizeof(addr)) < 0) - return -errno; - - return 0; -} - -static int get_family_id(int fd, const char *name) -{ - struct { - struct nlmsghdr nlh; - struct genlmsghdr genl; - char buf[256]; - } req = { 0 }; - char resp[8192]; - struct nlmsghdr *nlh; - struct genlmsghdr *genl; - struct nlattr *na; - int len; - int rem; - int ret; - - req.nlh.nlmsg_len = NLMSG_LENGTH(GENL_HDRLEN); - req.nlh.nlmsg_type = GENL_ID_CTRL; - req.nlh.nlmsg_flags = NLM_F_REQUEST; - req.nlh.nlmsg_seq = 1; - req.nlh.nlmsg_pid = getpid(); - - req.genl.cmd = CTRL_CMD_GETFAMILY; - req.genl.version = 1; - - na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len)); - na->nla_type = CTRL_ATTR_FAMILY_NAME; - na->nla_len = NLA_HDRLEN + strlen(name) + 1; - memcpy(taskstats_nla_data(na), name, strlen(name) + 1); - req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len); - - ret = send_request(fd, &req, req.nlh.nlmsg_len); - if (ret) - return ret; - - len = recv(fd, resp, sizeof(resp), 0); - if (len < 0) - return -errno; - - for (nlh = (struct nlmsghdr *)resp; NLMSG_OK(nlh, len); - nlh = NLMSG_NEXT(nlh, len)) { - if (nlh->nlmsg_type == NLMSG_ERROR) { - struct nlmsgerr *err = NLMSG_DATA(nlh); - - return err->error ? err->error : -ENOENT; - } - - genl = (struct genlmsghdr *)NLMSG_DATA(nlh); - rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; - na = (struct nlattr *)((char *)genl + GENL_HDRLEN); - while (taskstats_nla_ok(na, rem)) { - if (na->nla_type == CTRL_ATTR_FAMILY_ID) - return *(uint16_t *)taskstats_nla_data(na); - na = taskstats_nla_next(na, &rem); - } - } - - return -ENOENT; -} - static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id, struct taskstats *stats) { @@ -209,7 +91,7 @@ static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id, na = (struct nlattr *)((char *)&req + NLMSG_ALIGN(req.nlh.nlmsg_len)); na->nla_type = attr_type; na->nla_len = NLA_HDRLEN + sizeof(id); - memcpy(taskstats_nla_data(na), &id, sizeof(id)); + memcpy(nla_data(na), &id, sizeof(id)); req.nlh.nlmsg_len = NLMSG_ALIGN(req.nlh.nlmsg_len) + NLA_ALIGN(na->nla_len); ret = send_request(fd, &req, req.nlh.nlmsg_len); @@ -231,21 +113,21 @@ static int get_taskstats(int fd, int family_id, uint16_t attr_type, uint32_t id, genl = (struct genlmsghdr *)NLMSG_DATA(nlh); rem = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; na = (struct nlattr *)((char *)genl + GENL_HDRLEN); - while (taskstats_nla_ok(na, rem)) { + while (nla_ok(na, rem)) { if (na->nla_type == TASKSTATS_TYPE_AGGR_PID || na->nla_type == TASKSTATS_TYPE_AGGR_TGID) { - nested = (struct nlattr *)taskstats_nla_data(na); + nested = (struct nlattr *)nla_data(na); nrem = na->nla_len - NLA_HDRLEN; - while (taskstats_nla_ok(nested, nrem)) { + while (nla_ok(nested, nrem)) { if (nested->nla_type == TASKSTATS_TYPE_STATS) { - memcpy(stats, taskstats_nla_data(nested), + memcpy(stats, nla_data(nested), sizeof(*stats)); return 0; } - nested = taskstats_nla_next(nested, &nrem); + nested = nla_next(nested, &nrem); } } - na = taskstats_nla_next(na, &rem); + na = nla_next(na, &rem); } } diff --git a/tools/testing/selftests/ipc/msgque.c b/tools/testing/selftests/ipc/msgque.c index 82f73cdae120..4b4c1abfac0f 100644 --- a/tools/testing/selftests/ipc/msgque.c +++ b/tools/testing/selftests/ipc/msgque.c @@ -161,7 +161,7 @@ int dump_queue(struct msgque_data *msgque) ret = msgrcv(msgque->msq_id, &msgque->messages[i].mtype, MAX_MSG_SIZE, i, IPC_NOWAIT | MSG_COPY); if (ret < 0) { - if (errno == ENOSYS) + if (errno == EOPNOTSUPP) ksft_exit_skip("MSG_COPY not supported\n"); ksft_test_result_fail("Failed to copy IPC message: %m (%d)\n", errno); diff --git a/tools/testing/selftests/prctl/set-anon-vma-name-test.c b/tools/testing/selftests/prctl/set-anon-vma-name-test.c index ac6721b184a6..5f9589534da8 100644 --- a/tools/testing/selftests/prctl/set-anon-vma-name-test.c +++ b/tools/testing/selftests/prctl/set-anon-vma-name-test.c @@ -4,11 +4,13 @@ */ #include <errno.h> +#include <fcntl.h> #include <sys/prctl.h> #include <stdio.h> #include <stdlib.h> #include <sys/mman.h> #include <string.h> +#include <unistd.h> #include "kselftest_harness.h" @@ -73,15 +75,24 @@ int was_renaming_successful(char *target_name, unsigned long ptr) FIXTURE(vma) { void *ptr_anon, *ptr_not_anon; + int fd_not_anon; }; FIXTURE_SETUP(vma) { + char template[] = "./set-anon-vma-test-XXXXXX"; + self->ptr_anon = mmap(NULL, AREA_SIZE, PROT_READ | PROT_WRITE, - MAP_PRIVATE | MAP_ANONYMOUS, 0, 0); - ASSERT_NE(self->ptr_anon, NULL); + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + ASSERT_NE(self->ptr_anon, MAP_FAILED); + + self->fd_not_anon = mkstemp(template); + ASSERT_NE(self->fd_not_anon, -1); + unlink(template); + ASSERT_EQ(ftruncate(self->fd_not_anon, AREA_SIZE), 0); self->ptr_not_anon = mmap(NULL, AREA_SIZE, PROT_READ | PROT_WRITE, - MAP_PRIVATE, 0, 0); - ASSERT_NE(self->ptr_not_anon, NULL); + MAP_PRIVATE, self->fd_not_anon, 0); + ASSERT_NE(self->ptr_not_anon, MAP_FAILED); + close(self->fd_not_anon); } FIXTURE_TEARDOWN(vma) { @@ -98,7 +109,7 @@ TEST_F(vma, renaming) { EXPECT_EQ(rename_vma((unsigned long)self->ptr_anon, AREA_SIZE, BAD_NAME), -EINVAL); TH_LOG("Try to rename non-anonymous VMA"); - EXPECT_EQ(rename_vma((unsigned long) self->ptr_not_anon, AREA_SIZE, GOOD_NAME), -EINVAL); + EXPECT_EQ(rename_vma((unsigned long) self->ptr_not_anon, AREA_SIZE, GOOD_NAME), -EBADF); } TEST_HARNESS_MAIN |
