summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorPuranjay Mohan <puranjay@kernel.org>2026-09-22 13:00:52 -0700
committerAlexei Starovoitov <ast@kernel.org>2026-09-22 23:44:24 +0000
commitdb1b20ed6b093d98bfc3fb7ced2dce971d4e81ce (patch)
tree3c201dae36876f2058b6f805363abf3286a99779
parent908a60853b8dfe3bc9ad33598330bdcd03489b7c (diff)
downloadlinux-next-db1b20ed6b093d98bfc3fb7ced2dce971d4e81ce.tar.gz
linux-next-db1b20ed6b093d98bfc3fb7ced2dce971d4e81ce.zip
bpf: Add bpf_call_rcu_tasks_trace() kfunc
Sleepable BPF programs hold rcu_read_lock_trace(), not rcu_read_lock(), so a plain RCU grace period does not wait for them. A program whose readers are sleepable needs this flavour to defer reclaim safely. call_rcu_tasks_trace() is call_srcu() on rcu_tasks_trace_srcu_struct and SRCU invokes callbacks with BH disabled, so the callback is still not sleepable. It does run from a kworker rather than softirq or the rcuc/rcuo kthread, so a callback must not assume anything about current. Only the queueing call differs, so the two share struct bpf_rcu_head and all of the verifier plumbing. Signed-off-by: Puranjay Mohan <puranjay@kernel.org> Signed-off-by: Alexei Starovoitov <ast@kernel.org> Link: https://patch.msgid.link/20260922200208.3203834-4-puranjay@kernel.org
-rw-r--r--kernel/bpf/helpers.c57
-rw-r--r--kernel/bpf/verifier.c7
2 files changed, 47 insertions, 17 deletions
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c
index 8a01dd4058a0..301b35bd85c8 100644
--- a/kernel/bpf/helpers.c
+++ b/kernel/bpf/helpers.c
@@ -4838,21 +4838,10 @@ static void bpf_rcu_run_callback(struct rcu_head *rcu)
bpf_prog_put(prog);
}
-/**
- * bpf_call_rcu - Invoke a BPF callback after an RCU grace period
- * @rh: struct bpf_rcu_head in a BPF map value
- * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values
- * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh
- * @aux: bpf_prog_aux of the caller, implicitly set by the verifier
- *
- * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process
- * nor bpffs, or -EBADF if the calling program is going away.
- */
-__bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map,
- bpf_rcu_callback_t callback, struct bpf_prog_aux *aux)
+static int __bpf_call_rcu(struct bpf_rcu_head *rh, struct bpf_map *map, void *callback,
+ struct bpf_prog_aux *aux, bool trace)
{
struct bpf_rcu_head_kern *rhk = (void *)rh;
- struct bpf_map *map = map__const_map;
struct bpf_prog *prog;
BUILD_BUG_ON(sizeof(struct bpf_rcu_head_kern) > sizeof(struct bpf_rcu_head));
@@ -4872,13 +4861,50 @@ __bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map,
return -EBADF;
}
- rhk->callback_fn = (bpf_callback_t)(void *)callback;
+ rhk->callback_fn = (bpf_callback_t)callback;
rhk->map = map;
rhk->prog = prog;
- call_rcu(&rhk->rcu, bpf_rcu_run_callback);
+ if (trace)
+ call_rcu_tasks_trace(&rhk->rcu, bpf_rcu_run_callback);
+ else
+ call_rcu(&rhk->rcu, bpf_rcu_run_callback);
return 0;
}
+/**
+ * bpf_call_rcu - Invoke a BPF callback after an RCU grace period
+ * @rh: struct bpf_rcu_head in a BPF map value
+ * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values
+ * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh
+ * @aux: bpf_prog_aux of the caller, implicitly set by the verifier
+ *
+ * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process
+ * nor bpffs, or -EBADF if the calling program is going away.
+ */
+__bpf_kfunc int bpf_call_rcu(struct bpf_rcu_head *rh, void *map__const_map,
+ bpf_rcu_callback_t callback, struct bpf_prog_aux *aux)
+{
+ return __bpf_call_rcu(rh, map__const_map, callback, aux, false);
+}
+
+/**
+ * bpf_call_rcu_tasks_trace - Invoke a BPF callback after an RCU tasks trace grace period
+ * @rh: struct bpf_rcu_head in a BPF map value
+ * @map__const_map: bpf_map that embeds struct bpf_rcu_head in the values
+ * @callback: BPF subprogram, invoked as callback(map, key, value) for the value holding @rh
+ * @aux: bpf_prog_aux of the caller, implicitly set by the verifier
+ *
+ * Waits for sleepable BPF programs too. The callback itself is not sleepable either way.
+ *
+ * Return: 0, -EBUSY if @rh is already queued, -EPERM if @map is held by neither a process
+ * nor bpffs, or -EBADF if the calling program is going away.
+ */
+__bpf_kfunc int bpf_call_rcu_tasks_trace(struct bpf_rcu_head *rh, void *map__const_map,
+ bpf_rcu_callback_t callback, struct bpf_prog_aux *aux)
+{
+ return __bpf_call_rcu(rh, map__const_map, callback, aux, true);
+}
+
static int make_file_dynptr(struct file *file, u32 flags, bool may_sleep,
struct bpf_dynptr_kern *ptr)
{
@@ -5175,6 +5201,7 @@ BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE)
BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_call_rcu, KF_IMPLICIT_ARGS)
+BTF_ID_FLAGS(func, bpf_call_rcu_tasks_trace, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_dynptr_from_file)
BTF_ID_FLAGS(func, bpf_dynptr_file_discard, KF_RELEASE)
BTF_ID_FLAGS(func, bpf_timer_cancel_async)
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 64db47964ff9..a7c9e2d8965d 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -578,7 +578,7 @@ static bool is_async_cb_sleepable(struct bpf_verifier_env *env, struct bpf_insn
if (bpf_helper_call(insn) && insn->imm == BPF_FUNC_timer_set_callback)
return false;
- /* bpf_call_rcu callbacks are never sleepable. */
+ /* bpf_call_rcu and bpf_call_rcu_tasks_trace callbacks are never sleepable. */
if (bpf_pseudo_kfunc_call(insn) && insn->off == 0 && is_call_rcu_kfunc(insn->imm))
return false;
@@ -12629,6 +12629,7 @@ enum special_kfunc_type {
KF_bpf_task_work_schedule_signal,
KF_bpf_task_work_schedule_resume,
KF_bpf_call_rcu,
+ KF_bpf_call_rcu_tasks_trace,
KF_bpf_arena_alloc_pages,
KF_bpf_arena_free_pages,
KF_bpf_arena_reserve_pages,
@@ -12723,6 +12724,7 @@ BTF_ID(func, __bpf_trap)
BTF_ID(func, bpf_task_work_schedule_signal)
BTF_ID(func, bpf_task_work_schedule_resume)
BTF_ID(func, bpf_call_rcu)
+BTF_ID(func, bpf_call_rcu_tasks_trace)
BTF_ID(func, bpf_arena_alloc_pages)
BTF_ID(func, bpf_arena_free_pages)
BTF_ID(func, bpf_arena_reserve_pages)
@@ -12796,7 +12798,8 @@ static bool is_bpf_rbtree_add_kfunc(u32 func_id)
static bool is_call_rcu_kfunc(u32 func_id)
{
- return func_id == special_kfunc_list[KF_bpf_call_rcu];
+ return func_id == special_kfunc_list[KF_bpf_call_rcu] ||
+ func_id == special_kfunc_list[KF_bpf_call_rcu_tasks_trace];
}
static bool is_task_work_add_kfunc(u32 func_id)