diff options
| author | Puranjay Mohan <puranjay@kernel.org> | 2026-09-22 13:00:52 -0700 |
|---|---|---|
| committer | Alexei Starovoitov <ast@kernel.org> | 2026-09-22 23:44:24 +0000 |
| commit | db1b20ed6b093d98bfc3fb7ced2dce971d4e81ce (patch) | |
| tree | 3c201dae36876f2058b6f805363abf3286a99779 | |
| parent | 908a60853b8dfe3bc9ad33598330bdcd03489b7c (diff) | |
| download | linux-next-db1b20ed6b093d98bfc3fb7ced2dce971d4e81ce.tar.gz linux-next-db1b20ed6b093d98bfc3fb7ced2dce971d4e81ce.zip | |
bpf: Add bpf_call_rcu_tasks_trace() kfunc
Sleepable BPF programs hold rcu_read_lock_trace(), not rcu_read_lock(),
so a plain RCU grace period does not wait for them. A program whose
readers are sleepable needs this flavour to defer reclaim safely.
call_rcu_tasks_trace() is call_srcu() on rcu_tasks_trace_srcu_struct and
SRCU invokes callbacks with BH disabled, so the callback is still not
sleepable. It does run from a kworker rather than softirq or the
rcuc/rcuo kthread, so a callback must not assume anything about current.
Only the queueing call differs, so the two share struct bpf_rcu_head and
all of the verifier plumbing.
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
Signed-off-by: Alexei Starovoitov <ast@kernel.org>
Link: https://patch.msgid.link/20260922200208.3203834-4-puranjay@kernel.org
| -rw-r--r-- | kernel/bpf/helpers.c | 57 | ||||
| -rw-r--r-- | kernel/bpf/verifier.c | 7 |
2 files changed, 47 insertions, 17 deletions
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c index 8a01dd4058a0..301b35bd85c8 100644 --- a/kernel/bpf/helpers.c +++ b/kernel/bpf/helpers.c @@ -4838,21 +4838,10 @@ static void bpf_rcu_run_callback(struct rcu_head *rcu) bpf_prog_put(prog); } -/** - * bpf_call_rcu - Invoke a BPF callback after an RCU grace period - * @rh: struct bpf_rcu_head in a BPF map value - * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values - * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh - * @aux: bpf_prog_aux of the caller, implicitly set by the verifier - * - * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process - * nor bpffs, or -EBADF if the calling program is going away. - */ -__bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map, - bpf_rcu_callback_t callback, struct bpf_prog_aux *aux) +static int __bpf_call_rcu(struct bpf_rcu_head *rh, struct bpf_map *map, void *callback, + struct bpf_prog_aux *aux, bool trace) { struct bpf_rcu_head_kern *rhk = (void *)rh; - struct bpf_map *map = map__const_map; struct bpf_prog *prog; BUILD_BUG_ON(sizeof(struct bpf_rcu_head_kern) > sizeof(struct bpf_rcu_head)); @@ -4872,13 +4861,50 @@ __bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map, return -EBADF; } - rhk->callback_fn = (bpf_callback_t)(void *)callback; + rhk->callback_fn = (bpf_callback_t)callback; rhk->map = map; rhk->prog = prog; - call_rcu(&rhk->rcu, bpf_rcu_run_callback); + if (trace) + call_rcu_tasks_trace(&rhk->rcu, bpf_rcu_run_callback); + else + call_rcu(&rhk->rcu, bpf_rcu_run_callback); return 0; } +/** + * bpf_call_rcu - Invoke a BPF callback after an RCU grace period + * @rh: struct bpf_rcu_head in a BPF map value + * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values + * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh + * @aux: bpf_prog_aux of the caller, implicitly set by the verifier + * + * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process + * nor bpffs, or -EBADF if the calling program is going away. + */ +__bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map, + bpf_rcu_callback_t callback, struct bpf_prog_aux *aux) +{ + return __bpf_call_rcu(rh, map__const_map, callback, aux, false); +} + +/** + * bpf_call_rcu_tasks_trace - Invoke a BPF callback after an RCU tasks trace grace period + * @rh: struct bpf_rcu_head in a BPF map value + * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values + * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh + * @aux: bpf_prog_aux of the caller, implicitly set by the verifier + * + * Waits for sleepable BPF programs too. The callback itself is not sleepable either way. + * + * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process + * nor bpffs, or -EBADF if the calling program is going away. + */ +__bpf_kfunc int bpf_call_rcu_tasks_trace(struct bpf_rcu_head *rh, void *map__const_map, + bpf_rcu_callback_t callback, struct bpf_prog_aux *aux) +{ + return __bpf_call_rcu(rh, map__const_map, callback, aux, true); +} + static int make_file_dynptr(struct file *file, u32 flags, bool may_sleep, struct bpf_dynptr_kern *ptr) { @@ -5175,6 +5201,7 @@ BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE) BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_call_rcu, KF_IMPLICIT_ARGS) +BTF_ID_FLAGS(func, bpf_call_rcu_tasks_trace, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_dynptr_from_file) BTF_ID_FLAGS(func, bpf_dynptr_file_discard, KF_RELEASE) BTF_ID_FLAGS(func, bpf_timer_cancel_async) diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c index 64db47964ff9..a7c9e2d8965d 100644 --- a/kernel/bpf/verifier.c +++ b/kernel/bpf/verifier.c @@ -578,7 +578,7 @@ static bool is_async_cb_sleepable(struct bpf_verifier_env *env, struct bpf_insn if (bpf_helper_call(insn) && insn->imm == BPF_FUNC_timer_set_callback) return false; - /* bpf_call_rcu callbacks are never sleepable. */ + /* bpf_call_rcu and bpf_call_rcu_tasks_trace callbacks are never sleepable. */ if (bpf_pseudo_kfunc_call(insn) && insn->off == 0 && is_call_rcu_kfunc(insn->imm)) return false; @@ -12629,6 +12629,7 @@ enum special_kfunc_type { KF_bpf_task_work_schedule_signal, KF_bpf_task_work_schedule_resume, KF_bpf_call_rcu, + KF_bpf_call_rcu_tasks_trace, KF_bpf_arena_alloc_pages, KF_bpf_arena_free_pages, KF_bpf_arena_reserve_pages, @@ -12723,6 +12724,7 @@ BTF_ID(func, __bpf_trap) BTF_ID(func, bpf_task_work_schedule_signal) BTF_ID(func, bpf_task_work_schedule_resume) BTF_ID(func, bpf_call_rcu) +BTF_ID(func, bpf_call_rcu_tasks_trace) BTF_ID(func, bpf_arena_alloc_pages) BTF_ID(func, bpf_arena_free_pages) BTF_ID(func, bpf_arena_reserve_pages) @@ -12796,7 +12798,8 @@ static bool is_bpf_rbtree_add_kfunc(u32 func_id) static bool is_call_rcu_kfunc(u32 func_id) { - return func_id == special_kfunc_list[KF_bpf_call_rcu]; + return func_id == special_kfunc_list[KF_bpf_call_rcu] || + func_id == special_kfunc_list[KF_bpf_call_rcu_tasks_trace]; } static bool is_task_work_add_kfunc(u32 func_id) |
