summaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
Diffstat (limited to 'kernel')
-rw-r--r--kernel/bpf/btf.c136
-rw-r--r--kernel/bpf/check_btf.c12
-rw-r--r--kernel/bpf/core.c21
-rw-r--r--kernel/bpf/crypto.c5
-rw-r--r--kernel/bpf/fixups.c24
-rw-r--r--kernel/bpf/hashtab.c43
-rw-r--r--kernel/bpf/helpers.c60
-rw-r--r--kernel/bpf/memalloc.c50
-rw-r--r--kernel/bpf/offload.c2
-rw-r--r--kernel/bpf/states.c16
-rw-r--r--kernel/bpf/syscall.c15
-rw-r--r--kernel/bpf/verifier.c162
-rw-r--r--kernel/cgroup/cpuset.c3
-rw-r--r--kernel/cgroup/pids.c5
-rw-r--r--kernel/events/core.c17
-rw-r--r--kernel/exit.c28
-rw-r--r--kernel/fork.c2
-rw-r--r--kernel/kprobes.c22
-rw-r--r--kernel/power/hibernate.c26
-rw-r--r--kernel/sched/core.c2
-rw-r--r--kernel/sched/ext/cid.c26
-rw-r--r--kernel/sched/ext/ext.c156
-rw-r--r--kernel/sched/ext/inlines.h4
-rw-r--r--kernel/sched/ext/internal.h60
-rw-r--r--kernel/sched/ext/sub.c11
-rw-r--r--kernel/sched/ext/types.h4
-rw-r--r--kernel/sched/fair.c417
-rw-r--r--kernel/sched/topology.c22
-rw-r--r--kernel/trace/fprobe.c15
-rw-r--r--kernel/workqueue.c2
30 files changed, 963 insertions, 405 deletions
diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c
index 9f33e95d5741..d870bc5e50bc 100644
--- a/kernel/bpf/btf.c
+++ b/kernel/bpf/btf.c
@@ -4268,13 +4268,10 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec)
{
int i;
- /* There are three types that signify ownership of some other type:
- * kptr_ref, bpf_list_head, bpf_rb_root.
- * kptr_ref only supports storing kernel types, which can't store
- * references to program allocated local types.
- *
- * Hence we only need to ensure that bpf_{list_head,rb_root} ownership
- * does not form cycles.
+ /*
+ * Check fields which require the complete BTF and initialize runtime
+ * metadata. Ownership relationships are validated after every record has
+ * been fixed up.
*/
if (IS_ERR_OR_NULL(rec) || !(rec->field_mask & (BPF_GRAPH_ROOT | BPF_UPTR)))
return 0;
@@ -4305,51 +4302,88 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec)
if (!meta)
return -EFAULT;
rec->fields[i].graph_root.value_rec = meta->record;
+ }
+ return 0;
+}
- /* We need to set value_rec for all root types, but no need
- * to check ownership cycle for a type unless it's also a
- * node type.
- */
- if (!(rec->field_mask & BPF_GRAPH_NODE))
+static int btf_owned_type_idx(const struct btf *btf, struct btf_struct_metas *tab,
+ const struct btf_field *field)
+{
+ struct btf_struct_meta *meta;
+ u32 btf_id;
+
+ if (field->type & BPF_GRAPH_ROOT) {
+ btf_id = field->graph_root.value_btf_id;
+ } else if (field->type == BPF_KPTR_REF || field->type == BPF_KPTR_PERCPU) {
+ if (btf_is_kernel(field->kptr.btf))
+ return -ENOENT;
+ btf_id = field->kptr.btf_id;
+ } else {
+ return -ENOENT;
+ }
+
+ meta = btf_find_struct_meta(btf, btf_id);
+ if (!meta)
+ return field->type & BPF_GRAPH_ROOT ? -EFAULT : -ENOENT;
+ return meta - tab->types;
+}
+
+/*
+ * Each ownership edge adds kernel frames through bpf_obj_free_fields() and
+ * __bpf_obj_drop_impl(). Keep the bound deliberately small because object
+ * destruction can itself run below a BPF call chain. A final pointee without
+ * special fields is not present in the struct metadata table and adds only a
+ * non-recursing drop.
+ */
+#define BTF_MAX_OWNERSHIP_DEPTH 8
+
+static int btf_ownership_depth(const struct btf *btf,
+ struct btf_struct_metas *tab, u8 *depth,
+ int idx, int depth_left)
+{
+ const struct btf_record *rec = tab->types[idx].record;
+ int i, ret, max_depth = 0;
+
+ if (!depth_left)
+ return -ELOOP;
+ if (depth[idx])
+ goto done;
+
+ for (i = 0; i < rec->cnt; i++) {
+ ret = btf_owned_type_idx(btf, tab, &rec->fields[i]);
+ if (ret == -ENOENT)
continue;
+ if (ret < 0)
+ return ret;
+ ret = btf_ownership_depth(btf, tab, depth, ret, depth_left - 1);
+ if (ret < 0)
+ return ret;
+ max_depth = max(max_depth, ret);
+ }
+ depth[idx] = max_depth + 1;
+done:
+ return depth[idx] > depth_left ? -ELOOP : depth[idx];
+}
- /* We need to ensure ownership acyclicity among all types. The
- * proper way to do it would be to topologically sort all BTF
- * IDs based on the ownership edges, since there can be multiple
- * bpf_{list_head,rb_node} in a type. Instead, we use the
- * following resaoning:
- *
- * - A type can only be owned by another type in user BTF if it
- * has a bpf_{list,rb}_node. Let's call these node types.
- * - A type can only _own_ another type in user BTF if it has a
- * bpf_{list_head,rb_root}. Let's call these root types.
- *
- * We ensure that if a type is both a root and node, its
- * element types cannot be root types.
- *
- * To ensure acyclicity:
- *
- * When A is an root type but not a node, its ownership
- * chain can be:
- * A -> B -> C
- * Where:
- * - A is an root, e.g. has bpf_rb_root.
- * - B is both a root and node, e.g. has bpf_rb_node and
- * bpf_list_head.
- * - C is only an root, e.g. has bpf_list_node
- *
- * When A is both a root and node, some other type already
- * owns it in the BTF domain, hence it can not own
- * another root type through any of the ownership edges.
- * A -> B
- * Where:
- * - A is both an root and node.
- * - B is only an node.
- */
- if (meta->record->field_mask & BPF_GRAPH_ROOT)
- return -ELOOP;
+static int btf_check_ownership_depth(const struct btf *btf,
+ struct btf_struct_metas *tab)
+{
+ u8 *depth;
+ int i, ret = 0;
+
+ depth = kvcalloc(tab->cnt, sizeof(*depth), GFP_KERNEL | __GFP_NOWARN);
+ if (!depth)
+ return -ENOMEM;
+
+ for (i = 0; i < tab->cnt; i++) {
+ ret = btf_ownership_depth(btf, tab, depth, i,
+ BTF_MAX_OWNERSHIP_DEPTH);
+ if (ret < 0)
+ break;
+ ret = 0;
}
- return 0;
+ kvfree(depth);
+ return ret;
}
static void __btf_struct_show(const struct btf *btf, const struct btf_type *t,
@@ -6044,6 +6078,10 @@ static struct btf *btf_parse(const union bpf_attr *attr, bpfptr_t uattr,
if (err < 0)
goto errout_meta;
}
+
+ err = btf_check_ownership_depth(btf, struct_meta_tab);
+ if (err < 0)
+ goto errout_meta;
}
err = bpf_log_attr_finalize(attr_log, &env->log);
@@ -7187,7 +7225,7 @@ again:
if (btf_type_is_int(t))
return WALK_SCALAR;
- if (!btf_type_is_struct(t))
+ if (!btf_type_is_struct(t) || !t->size)
goto error;
off = (off - moff) % t->size;
diff --git a/kernel/bpf/check_btf.c b/kernel/bpf/check_btf.c
index 0e8b3ccc7a5b..4c1ed842f661 100644
--- a/kernel/bpf/check_btf.c
+++ b/kernel/bpf/check_btf.c
@@ -338,9 +338,9 @@ err_free:
#define MIN_CORE_RELO_SIZE sizeof(struct bpf_core_relo)
#define MAX_CORE_RELO_SIZE MAX_FUNCINFO_REC_SIZE
-static int check_core_relo(struct bpf_verifier_env *env,
- const union bpf_attr *attr,
- bpfptr_t uattr)
+int bpf_check_core_relo(struct bpf_verifier_env *env,
+ const union bpf_attr *attr,
+ bpfptr_t uattr)
{
u32 i, nr_core_relo, ncopy, expected_size, rec_size;
struct bpf_core_relo core_relo = {};
@@ -414,7 +414,7 @@ int bpf_prepare_btf_info(struct bpf_verifier_env *env,
struct btf *btf;
int err;
- if (!attr->func_info_cnt && !attr->line_info_cnt) {
+ if (!attr->func_info_cnt && !attr->line_info_cnt && !attr->core_relo_cnt) {
if (check_abnormal_return(env))
return -EINVAL;
return 0;
@@ -455,9 +455,5 @@ int bpf_check_btf_info(struct bpf_verifier_env *env,
if (err)
return err;
- err = check_core_relo(env, attr, uattr);
- if (err)
- return err;
-
return 0;
}
diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c
index 8b294dfc1ad4..2e3bf8113ae9 100644
--- a/kernel/bpf/core.c
+++ b/kernel/bpf/core.c
@@ -19,6 +19,7 @@
#include <uapi/linux/btf.h>
#include <linux/filter.h>
+#include <linux/sched/signal.h>
#include <linux/skbuff.h>
#include <linux/static_call.h>
#include <linux/vmalloc.h>
@@ -1619,6 +1620,8 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp
* fix it up here on error.
*/
bpf_jit_prog_release_other(prog, clone);
+ if (env && fatal_signal_pending(current))
+ return ERR_PTR(-EINTR);
return IS_ERR(tmp) ? tmp : ERR_PTR(-ENOMEM);
}
@@ -2636,11 +2639,14 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc
orig_prog = prog;
prog = bpf_jit_blind_constants(env, prog);
/*
- * If blinding was requested and we failed during blinding, we must fall
- * back to the interpreter.
+ * Fall back to the interpreter after blinding failures, except when
+ * the loader was killed.
*/
- if (IS_ERR(prog))
+ if (IS_ERR(prog)) {
+ if (PTR_ERR(prog) == -EINTR)
+ return prog;
goto out_restore;
+ }
prog = bpf_int_jit_compile(env, prog);
if (prog->jited) {
@@ -2659,6 +2665,8 @@ out_restore:
struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct bpf_prog *fp,
int *err)
{
+ struct bpf_prog *jit_prog;
+
/* In case of BPF to BPF calls, verifier did all the prep
* work with regards to JITing, etc.
*/
@@ -2681,7 +2689,12 @@ struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct
if (*err)
return fp;
- fp = bpf_prog_jit_compile(env, fp);
+ jit_prog = bpf_prog_jit_compile(env, fp);
+ if (IS_ERR(jit_prog)) {
+ *err = PTR_ERR(jit_prog);
+ return fp;
+ }
+ fp = jit_prog;
bpf_prog_jit_attempt_done(fp);
if (!fp->jited && jit_needed) {
*err = -ENOTSUPP;
diff --git a/kernel/bpf/crypto.c b/kernel/bpf/crypto.c
index 51f89cecefb4..3f3fe2450fc6 100644
--- a/kernel/bpf/crypto.c
+++ b/kernel/bpf/crypto.c
@@ -149,8 +149,9 @@ bpf_crypto_ctx_create(const struct bpf_crypto_params *params, u32 params__sz,
const struct bpf_crypto_type *type;
struct bpf_crypto_ctx *ctx;
- if (!params || params->reserved[0] || params->reserved[1] ||
- params__sz != sizeof(struct bpf_crypto_params)) {
+ if (!params ||
+ params__sz != sizeof(struct bpf_crypto_params) ||
+ params->reserved[0] || params->reserved[1]) {
*err = -EINVAL;
return NULL;
}
diff --git a/kernel/bpf/fixups.c b/kernel/bpf/fixups.c
index 52d3cec33672..d6f83521fc78 100644
--- a/kernel/bpf/fixups.c
+++ b/kernel/bpf/fixups.c
@@ -8,6 +8,7 @@
#include <linux/bsearch.h>
#include <linux/sort.h>
#include <linux/perf_event.h>
+#include <linux/sched/signal.h>
#include <net/xdp.h>
#include "disasm.h"
@@ -306,12 +307,28 @@ static void adjust_poke_descs(struct bpf_prog *prog, u32 off, u32 len)
}
}
+/*
+ * Some post-verification instruction rewriting passes require an
+ * O(prog->len) operation per instruction. Keep their shared primitives
+ * killable and preemptible.
+ */
+static bool bpf_rewrite_must_abort(void)
+{
+ if (fatal_signal_pending(current))
+ return true;
+ cond_resched();
+ return false;
+}
+
struct bpf_prog *bpf_patch_insn_data(struct bpf_verifier_env *env, u32 off,
const struct bpf_insn *patch, u32 len)
{
struct bpf_prog *new_prog;
struct bpf_insn_aux_data *new_data = NULL;
+ if (bpf_rewrite_must_abort())
+ return NULL;
+
if (len > 1) {
new_data = vrealloc(env->insn_aux_data,
array_size(env->prog->len + len - 1,
@@ -523,6 +540,9 @@ static int verifier_remove_insns(struct bpf_verifier_env *env, u32 off, u32 cnt)
unsigned int orig_prog_len = env->prog->len;
int err;
+ if (bpf_rewrite_must_abort())
+ return -EINTR;
+
if (bpf_prog_is_offloaded(env->prog->aux))
bpf_prog_offload_remove_insns(env, off, cnt);
@@ -1356,7 +1376,7 @@ int bpf_jit_subprogs(struct bpf_verifier_env *env)
}
prog = bpf_jit_blind_constants(env, prog);
if (IS_ERR(prog)) {
- err = -ENOMEM;
+ err = PTR_ERR(prog);
prog = orig_prog;
goto out_restore;
}
@@ -1433,7 +1453,7 @@ int bpf_fixup_call_args(struct bpf_verifier_env *env)
err = bpf_jit_subprogs(env);
if (err == 0)
return 0;
- if (err == -EFAULT)
+ if (err == -EFAULT || err == -EINTR)
return err;
}
#ifndef CONFIG_BPF_JIT_ALWAYS_ON
diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c
index 6f331c80130d..f9464e566f10 100644
--- a/kernel/bpf/hashtab.c
+++ b/kernel/bpf/hashtab.c
@@ -1054,14 +1054,17 @@ static void pcpu_init_value(struct bpf_htab *htab, void __percpu *pptr,
/* When not setting the initial value on all cpus, zero-fill element
* values for other cpus. Otherwise, bpf program has no way to ensure
* known initial values for cpus other than current one
- * (onallcpus=false always when coming from bpf prog).
+ * (onallcpus=false always when coming from bpf prog,
+ * map_flags & BPF_F_CPU when coming from syscall but setting
+ * only one cpu).
*/
- if (!onallcpus) {
- int current_cpu = raw_smp_processor_id();
+ if (!onallcpus || (map_flags & BPF_F_CPU)) {
+ int init_cpu = (map_flags & BPF_F_CPU) ? map_flags >> 32 :
+ raw_smp_processor_id();
int cpu;
for_each_possible_cpu(cpu) {
- if (cpu == current_cpu)
+ if (cpu == init_cpu)
copy_map_value(&htab->map, per_cpu_ptr(pptr, cpu), value);
else /* Since elem is preallocated, we cannot touch special fields */
zero_map_value(&htab->map, per_cpu_ptr(pptr, cpu));
@@ -1772,6 +1775,12 @@ static int htab_lru_percpu_map_lookup_and_delete_elem(struct bpf_map *map,
flags);
}
+/*
+ * Max consecutive empty buckets to walk in one RCU +
+ * instrumentation-disabled section before rescheduling.
+ */
+#define HTAB_BATCH_EMPTY_RESCHED 64
+
static int
__htab_map_lookup_and_delete_batch(struct bpf_map *map,
const union bpf_attr *attr,
@@ -1793,6 +1802,7 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map,
unsigned long flags = 0;
bool locked = false;
struct htab_elem *l;
+ u32 empty_cnt = 0;
struct bucket *b;
int ret = 0;
@@ -1971,30 +1981,41 @@ again_nocopy:
}
next_batch:
- /* If we are not copying data, we can go to next bucket and avoid
- * unlocking the rcu.
+ /*
+ * If we are not copying data, we can go to next bucket and avoid
+ * unlocking the rcu. Bound the walk though: after
+ * HTAB_BATCH_EMPTY_RESCHED consecutive empty buckets, fully exit
+ * the critical section (no locks are held here) and reschedule.
*/
if (!bucket_cnt && (batch + 1 < htab->n_buckets)) {
batch++;
- goto again_nocopy;
+ if (++empty_cnt < HTAB_BATCH_EMPTY_RESCHED)
+ goto again_nocopy;
+ empty_cnt = 0;
+ rcu_read_unlock();
+ bpf_enable_instrumentation();
+ cond_resched_tasks_rcu_qs();
+ goto again;
}
rcu_read_unlock();
bpf_enable_instrumentation();
- if (bucket_cnt && (copy_to_user(ukeys + total * key_size, keys,
- key_size * bucket_cnt) ||
- copy_to_user(uvalues + total * value_size, values,
- value_size * bucket_cnt))) {
+ if (bucket_cnt && (copy_to_user(ukeys + (size_t)total * key_size, keys,
+ (size_t)key_size * bucket_cnt) ||
+ copy_to_user(uvalues + (size_t)total * value_size, values,
+ (size_t)value_size * bucket_cnt))) {
ret = -EFAULT;
goto after_loop;
}
total += bucket_cnt;
+ empty_cnt = 0;
batch++;
if (batch >= htab->n_buckets) {
ret = -ENOENT;
goto after_loop;
}
+ cond_resched_tasks_rcu_qs();
goto again;
after_loop:
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c
index b3cc5c8fc875..712dca5a2c5b 100644
--- a/kernel/bpf/helpers.c
+++ b/kernel/bpf/helpers.c
@@ -4883,7 +4883,7 @@ BTF_ID(func, bpf_cgroup_release_dtor)
BTF_KFUNCS_START(common_btf_ids)
BTF_ID_FLAGS(func, bpf_cast_to_kern_ctx, KF_FASTCALL)
-BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL)
+BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_rcu_read_lock)
BTF_ID_FLAGS(func, bpf_rcu_read_unlock)
BTF_ID_FLAGS(func, bpf_dynptr_slice, KF_RET_NULL)
@@ -4920,26 +4920,26 @@ BTF_ID_FLAGS(func, bpf_wq_set_callback, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_wq_start)
BTF_ID_FLAGS(func, bpf_preempt_disable)
BTF_ID_FLAGS(func, bpf_preempt_enable)
-BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW)
+BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_iter_bits_next, KF_ITER_NEXT | KF_RET_NULL)
BTF_ID_FLAGS(func, bpf_iter_bits_destroy, KF_ITER_DESTROY)
-BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE)
-BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE)
-BTF_ID_FLAGS(func, bpf_get_kmem_cache)
+BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_get_kmem_cache, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_new, KF_ITER_NEW | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_local_irq_save)
BTF_ID_FLAGS(func, bpf_local_irq_restore)
#ifdef CONFIG_BPF_EVENTS
-BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr)
-BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr)
-BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr)
-BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr)
-BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE)
-BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE)
-BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE)
-BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr, KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr, KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr, KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr, KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE | KF_PERFMON)
#endif
#ifdef CONFIG_DMA_SHARED_BUFFER
BTF_ID_FLAGS(func, bpf_iter_dmabuf_new, KF_ITER_NEW | KF_SLEEPABLE)
@@ -4947,26 +4947,26 @@ BTF_ID_FLAGS(func, bpf_iter_dmabuf_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPAB
BTF_ID_FLAGS(func, bpf_iter_dmabuf_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
#endif
BTF_ID_FLAGS(func, __bpf_trap)
-BTF_ID_FLAGS(func, bpf_strcmp);
-BTF_ID_FLAGS(func, bpf_strcasecmp);
-BTF_ID_FLAGS(func, bpf_strncasecmp);
-BTF_ID_FLAGS(func, bpf_strchr);
-BTF_ID_FLAGS(func, bpf_strchrnul);
-BTF_ID_FLAGS(func, bpf_strnchr);
-BTF_ID_FLAGS(func, bpf_strrchr);
-BTF_ID_FLAGS(func, bpf_strlen);
-BTF_ID_FLAGS(func, bpf_strnlen);
-BTF_ID_FLAGS(func, bpf_strspn);
-BTF_ID_FLAGS(func, bpf_strcspn);
-BTF_ID_FLAGS(func, bpf_strstr);
-BTF_ID_FLAGS(func, bpf_strcasestr);
-BTF_ID_FLAGS(func, bpf_strnstr);
-BTF_ID_FLAGS(func, bpf_strncasestr);
+BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strcasecmp, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strncasecmp, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strchr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strchrnul, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strnchr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strrchr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strlen, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strnlen, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strspn, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strcspn, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strstr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strcasestr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strnstr, KF_PERFMON);
+BTF_ID_FLAGS(func, bpf_strncasestr, KF_PERFMON);
#if defined(CONFIG_BPF_LSM) && defined(CONFIG_CGROUPS)
BTF_ID_FLAGS(func, bpf_cgroup_read_xattr, KF_RCU)
#endif
-BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE)
-BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE)
+BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON)
+BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_dynptr_from_file)
diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c
index e9662db7198f..8a8f088e83e6 100644
--- a/kernel/bpf/memalloc.c
+++ b/kernel/bpf/memalloc.c
@@ -119,6 +119,7 @@ struct bpf_mem_cache {
struct llist_head waiting_for_gp_ttrace;
struct rcu_head rcu_ttrace;
atomic_t call_rcu_ttrace_in_progress;
+ raw_spinlock_t lock;
};
struct bpf_mem_caches {
@@ -214,25 +215,24 @@ static void alloc_bulk(struct bpf_mem_cache *c, int cnt, int node, bool atomic)
gfp = __GFP_NOWARN | __GFP_ACCOUNT;
gfp |= atomic ? GFP_NOWAIT : GFP_KERNEL;
- for (i = 0; i < cnt; i++) {
- /*
- * For every 'c' llist_del_first(&c->free_by_rcu_ttrace); is
- * done only by one CPU == current CPU. Other CPUs might
- * llist_add() and llist_del_all() in parallel.
- */
- obj = llist_del_first(&c->free_by_rcu_ttrace);
- if (!obj)
- break;
- add_obj_to_free_list(c, obj);
- }
- if (i >= cnt)
- return;
+ /*
+ * c->lock serializes concurrent llist_del_first() against
+ * llist_del_all() in __free_rcu() and do_call_rcu_ttrace().
+ */
+ scoped_guard(raw_spinlock_irqsave, &c->lock) {
+ for (i = 0; i < cnt; i++) {
+ obj = llist_del_first(&c->free_by_rcu_ttrace);
+ if (!obj)
+ break;
+ add_obj_to_free_list(c, obj);
+ }
- for (; i < cnt; i++) {
- obj = llist_del_first(&c->waiting_for_gp_ttrace);
- if (!obj)
- break;
- add_obj_to_free_list(c, obj);
+ for (; i < cnt; i++) {
+ obj = llist_del_first(&c->waiting_for_gp_ttrace);
+ if (!obj)
+ break;
+ add_obj_to_free_list(c, obj);
+ }
}
if (i >= cnt)
return;
@@ -279,8 +279,12 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per
static void __free_rcu(struct rcu_head *head)
{
struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace);
+ struct llist_node *llnode;
+
+ scoped_guard(raw_spinlock_irqsave, &c->lock)
+ llnode = llist_del_all(&c->waiting_for_gp_ttrace);
- free_all(c, llist_del_all(&c->waiting_for_gp_ttrace), !!c->percpu_size);
+ free_all(c, llnode, !!c->percpu_size);
atomic_set(&c->call_rcu_ttrace_in_progress, 0);
}
@@ -300,7 +304,8 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) {
if (unlikely(READ_ONCE(c->draining))) {
- llnode = llist_del_all(&c->free_by_rcu_ttrace);
+ scoped_guard(raw_spinlock_irqsave, &c->lock)
+ llnode = llist_del_all(&c->free_by_rcu_ttrace);
free_all(c, llnode, !!c->percpu_size);
}
return;
@@ -535,6 +540,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
+ raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}
@@ -557,7 +563,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
-
+ raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}
@@ -609,7 +615,7 @@ int bpf_mem_alloc_percpu_unit_init(struct bpf_mem_alloc *ma, int size)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
-
+ raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}
diff --git a/kernel/bpf/offload.c b/kernel/bpf/offload.c
index 0d6f5569588c..d855399812ee 100644
--- a/kernel/bpf/offload.c
+++ b/kernel/bpf/offload.c
@@ -698,6 +698,8 @@ static bool __bpf_offload_dev_match(struct bpf_prog *prog,
return false;
if (offload->netdev == netdev)
return true;
+ if (!bpf_prog_is_offloaded(prog->aux))
+ return false;
ondev1 = bpf_offload_find_netdev(offload->netdev);
ondev2 = bpf_offload_find_netdev(netdev);
diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c
index 66fb11b6c6a7..012b82513a3b 100644
--- a/kernel/bpf/states.c
+++ b/kernel/bpf/states.c
@@ -491,7 +491,8 @@ static bool regs_exact(const struct bpf_reg_state *rold,
{
return memcmp(rold, rcur, offsetof(struct bpf_reg_state, id)) == 0 &&
check_ids(rold->id, rcur->id, idmap) &&
- check_ids(rold->parent_id, rcur->parent_id, idmap);
+ check_ids(rold->parent_id, rcur->parent_id, idmap) &&
+ check_ids(rold->map_uid, rcur->map_uid, idmap);
}
enum exact_level {
@@ -616,7 +617,8 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold,
range_within(rold, rcur) &&
tnum_in(rold->var_off, rcur->var_off) &&
check_ids(rold->id, rcur->id, idmap) &&
- check_ids(rold->parent_id, rcur->parent_id, idmap);
+ check_ids(rold->parent_id, rcur->parent_id, idmap) &&
+ check_ids(rold->map_uid, rcur->map_uid, idmap);
case PTR_TO_PACKET_META:
case PTR_TO_PACKET:
/* We must have at least as much range as the old ptr
@@ -635,14 +637,14 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold,
/* id relations must be preserved */
if (!check_ids(rold->id, rcur->id, idmap))
return false;
+ /* Preserve displacements between pointers sharing an ID. */
+ if (rold->id && rold->r64.base != rcur->r64.base)
+ return false;
/* new val must satisfy old val knowledge */
return range_within(rold, rcur) &&
tnum_in(rold->var_off, rcur->var_off);
case PTR_TO_STACK:
- /* two stack pointers are equal only if they're pointing to
- * the same stack frame, since fp-8 in foo != fp-8 in bar
- */
- return regs_exact(rold, rcur, idmap) && rold->frameno == rcur->frameno;
+ return regs_exact(rold, rcur, idmap);
case PTR_TO_ARENA:
return true;
case PTR_TO_INSN:
@@ -1121,7 +1123,7 @@ static bool states_maybe_looping(struct bpf_verifier_state *old,
fcur = cur->frame[fr];
for (i = 0; i < MAX_BPF_REG; i++)
if (memcmp(&fold->regs[i], &fcur->regs[i],
- offsetof(struct bpf_reg_state, frameno)))
+ offsetof(struct bpf_reg_state, precise)))
return false;
return true;
}
diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c
index c7bc9ba9b331..244a939b9d2d 100644
--- a/kernel/bpf/syscall.c
+++ b/kernel/bpf/syscall.c
@@ -2036,7 +2036,7 @@ int generic_map_delete_batch(struct bpf_map *map,
for (cp = 0; cp < max_count; cp++) {
err = -EFAULT;
- if (copy_from_user(key, keys + cp * map->key_size,
+ if (copy_from_user(key, keys + (size_t)cp * map->key_size,
map->key_size))
break;
@@ -2098,9 +2098,9 @@ int generic_map_update_batch(struct bpf_map *map, struct file *map_file,
for (cp = 0; cp < max_count; cp++) {
err = -EFAULT;
- if (copy_from_user(key, keys + cp * map->key_size,
+ if (copy_from_user(key, keys + (size_t)cp * map->key_size,
map->key_size) ||
- copy_from_user(value, values + cp * value_size, value_size))
+ copy_from_user(value, values + (size_t)cp * value_size, value_size))
break;
err = bpf_map_update_value(map, map_file, key, value,
@@ -2179,12 +2179,12 @@ int generic_map_lookup_batch(struct bpf_map *map,
if (err)
goto free_buf;
- if (copy_to_user(keys + cp * map->key_size, key,
+ if (copy_to_user(keys + (size_t)cp * map->key_size, key,
map->key_size)) {
err = -EFAULT;
goto free_buf;
}
- if (copy_to_user(values + cp * value_size, value, value_size)) {
+ if (copy_to_user(values + (size_t)cp * value_size, value, value_size)) {
err = -EFAULT;
goto free_buf;
}
@@ -6042,7 +6042,10 @@ struct bpf_link *bpf_link_get_curr_or_next(u32 *id)
again:
link = idr_get_next(&link_idr, id);
if (link) {
- link = bpf_link_inc_not_zero(link);
+ if (link->id)
+ link = bpf_link_inc_not_zero(link);
+ else
+ link = ERR_PTR(-EAGAIN);
if (IS_ERR(link)) {
(*id)++;
goto again;
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 72a3f5998dd2..41b49c56e123 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -567,7 +567,7 @@ static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_s
}
off = reg->var_off.value;
- if (off % BPF_REG_SIZE) {
+ if (off >= 0 || off % BPF_REG_SIZE) {
verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
return -EINVAL;
}
@@ -1864,6 +1864,7 @@ static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm)
offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type));
reg->id = 0;
reg->parent_id = 0;
+ reg->map_uid = 0;
___mark_reg_known(reg, imm);
}
@@ -1925,17 +1926,18 @@ static void refine_map_lookup_value(struct bpf_reg_state *reg)
if (map->inner_map_meta) {
reg->type = CONST_PTR_TO_MAP | maybe_null;
reg->map_ptr = map->inner_map_meta;
- /* transfer reg's id which is unique for every map_lookup_elem
+ /*
+ * transfer reg's id which is unique for every map_lookup_elem
* as UID of the inner map.
*/
- if (btf_record_has_field(map->inner_map_meta->record,
- BPF_TIMER | BPF_WORKQUEUE | BPF_TASK_WORK))
- reg->map_uid = reg->id;
+ reg->map_uid = reg->id;
} else if (map->map_type == BPF_MAP_TYPE_XSKMAP) {
reg->type = PTR_TO_XDP_SOCK | maybe_null;
+ reg->map_uid = 0;
} else if (map->map_type == BPF_MAP_TYPE_SOCKMAP ||
map->map_type == BPF_MAP_TYPE_SOCKHASH) {
reg->type = PTR_TO_SOCKET | maybe_null;
+ reg->map_uid = 0;
}
}
@@ -3042,6 +3044,8 @@ static int check_subprogs(struct bpf_verifier_env *env)
subprog[cur_subprog].exit_idx = i;
goto next;
}
+ if (insn_is_gotox(&insn[i]))
+ goto next;
off = i + bpf_jmp_offset(&insn[i]) + 1;
if (off < subprog_start || off >= subprog_end) {
verbose(env, "jump out of range from insn %d to %d\n", i, off);
@@ -3061,7 +3065,8 @@ next:
*/
if (code != (BPF_JMP | BPF_EXIT) &&
code != (BPF_JMP32 | BPF_JA) &&
- code != (BPF_JMP | BPF_JA)) {
+ code != (BPF_JMP | BPF_JA) &&
+ !insn_is_gotox(&insn[i])) {
verbose(env, "last insn is not an exit or jmp\n");
bpf_diag_program_structure(
env, i, "subprogram can fall through",
@@ -3582,7 +3587,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
save_register_state(env, state, spi, reg, size);
/* Break the relation on a narrowing spill. */
if (!reg_value_fits)
- state->stack[spi].spilled_ptr.id = 0;
+ clear_scalar_id(&state->stack[spi].spilled_ptr);
} else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
env->bpf_capable) {
struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
@@ -6453,6 +6458,15 @@ static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct b
return -EACCES;
}
+ if (rdonly_untrusted && !env->allow_ptr_leaks) {
+ verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
+ reg_type_str(env, reg->type));
+ bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory",
+ "the access requires CAP_PERFMON",
+ "Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers.");
+ return -EPERM;
+ }
+
/*
* Accesses to untrusted PTR_TO_MEM are done through probe
* instructions, hence no need to check bounds in that case.
@@ -9736,6 +9750,16 @@ static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog,
if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL,
NULL))
return -EINVAL;
+ /*
+ * PTR_TO_PACKET get passed as PTR_TO_MEM, preventing
+ * us from adjusting bounds tracking info.
+ */
+ if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) &&
+ sub->changes_pkt_data) {
+ bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n",
+ reg_arg_name(env, argno), subprog);
+ return -EINVAL;
+ }
if (!(arg->arg_type & PTR_MAYBE_NULL) &&
(type_may_be_null(reg->type) || bpf_register_is_null(reg))) {
bpf_log(log, "%s is expected to be non-NULL\n",
@@ -9919,6 +9943,7 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
if (err == -EFAULT)
return err;
if (bpf_subprog_is_global(env, subprog)) {
+ struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog);
const char *sub_name = bpf_subprog_name(env, subprog);
const char *operation;
bool returns_void;
@@ -9950,11 +9975,10 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
if (env->log.level & BPF_LOG_LEVEL)
verbose(env, "Func#%d ('%s') is global and assumed valid.\n",
subprog, sub_name);
+ sub_aux->called[in_sleepable_context(env)] = true;
returns_void = subprog_returns_void(env, subprog);
if (env->subprog_info[subprog].changes_pkt_data)
clear_all_pkt_pointers(env);
- /* mark global subprog for verifying after main prog */
- subprog_aux(env, subprog)->called = true;
if (returns_void)
bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
else
@@ -10036,6 +10060,7 @@ int map_set_for_each_callback_args(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr;
callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid;
+ callee->regs[BPF_REG_3].id = ++env->id_gen;
/* pointer to stack or null */
callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3];
@@ -10132,6 +10157,7 @@ static int set_timer_callback_state(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = map_ptr;
callee->regs[BPF_REG_3].map_uid = map_uid;
+ callee->regs[BPF_REG_3].id = ++env->id_gen;
/* unused */
bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
@@ -10250,6 +10276,7 @@ static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = map_ptr;
callee->regs[BPF_REG_3].map_uid = map_uid;
+ callee->regs[BPF_REG_3].id = ++env->id_gen;
/* unused */
bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
@@ -10772,11 +10799,7 @@ int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id,
/* Check if we're in a sleepable context. */
static inline bool in_sleepable_context(struct bpf_verifier_env *env)
{
- return !env->cur_state->active_rcu_locks &&
- !env->cur_state->active_preempt_locks &&
- !env->cur_state->active_locks &&
- !env->cur_state->active_irq_id &&
- in_sleepable(env);
+ return !in_rcu_cs(env);
}
static const char *non_sleepable_context_description(struct bpf_verifier_env *env)
@@ -11356,6 +11379,11 @@ static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta)
return meta->kfunc_flags & KF_DESTRUCTIVE;
}
+static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta)
+{
+ return meta->kfunc_flags & KF_PERFMON;
+}
+
static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta)
{
return meta->kfunc_flags & KF_RCU;
@@ -13834,6 +13862,15 @@ static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
return -EACCES;
}
+ if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) {
+ verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
+ func_name);
+ operation = bpf_diag_fmt(env, "kfunc %s", func_name);
+ bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON",
+ "Load the program with CAP_PERFMON, or avoid the kfunc.");
+ return -EPERM;
+ }
+
sleepable = bpf_is_kfunc_sleepable(&meta);
if (sleepable && !in_sleepable(env)) {
verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name);
@@ -15704,6 +15741,7 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg;
struct bpf_reg_state *ptr_reg = NULL, off_reg = {0};
bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
+ struct bpf_insn_aux_data *aux = cur_aux(env);
u8 opcode = BPF_OP(insn->code);
int err;
@@ -15715,13 +15753,24 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
/* Case where at least one operand is an arena. */
if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) {
- struct bpf_insn_aux_data *aux = cur_aux(env);
if (dst_reg->type != PTR_TO_ARENA)
*dst_reg = *src_reg;
if (BPF_CLASS(insn->code) == BPF_ALU64) {
/*
+ * Only arena pointers set needs_zext, but doing so
+ * modifies the instruction at fixup time to an ALU32
+ * and makes it unsuitable for 64-bit scalar args. We
+ * prevent zext from being set if the instruction has
+ * been previously called with non-arena registers.
+ */
+ if (aux->prevent_zext) {
+ verbose(env, "same insn cannot be used with and without arena pointer\n");
+ return -EINVAL;
+ }
+
+ /*
* 32-bit operations zero upper bits automatically.
* 64-bit operations need to be converted to 32.
*/
@@ -15733,6 +15782,16 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
return 0;
}
+ /* Prevent the instruction from being used with arena pointers (see above). */
+ if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) {
+ if (aux->needs_zext) {
+ verbose(env, "same insn cannot be used with and without arena pointer\n");
+ return -EINVAL;
+ }
+
+ aux->prevent_zext = true;
+ }
+
if (dst_reg->type != SCALAR_VALUE)
ptr_reg = dst_reg;
@@ -19421,13 +19480,14 @@ static void free_states(struct bpf_verifier_env *env)
}
}
-static int do_check_common(struct bpf_verifier_env *env, int subprog)
+static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable)
{
bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
struct bpf_subprog_info *sub = subprog_info(env, subprog);
struct bpf_prog_aux *aux = env->prog->aux;
struct bpf_verifier_state *state;
struct bpf_reg_state *regs;
+ u32 old_insns_total = sub->insns_total;
u32 insn_processed = env->insn_processed;
int ret, i;
@@ -19440,7 +19500,7 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog)
state->curframe = 0;
state->speculative = false;
state->branches = 1;
- state->in_sleepable = env->prog->sleepable;
+ state->in_sleepable = is_sleepable;
state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT);
if (!state->frame[0]) {
kfree(state);
@@ -19581,8 +19641,10 @@ out:
* not accounted as callees by account_current_path().
* Accumulate their total counts as total counts of the main or
* global subprog hosting the async call.
+ * Start from the saved total of earlier contexts: adding to the current
+ * total would count this pass's synchronous paths twice.
*/
- env->subprog_info[subprog].insns_total = env->insn_processed - insn_processed;
+ sub->insns_total = old_insns_total + (env->insn_processed - insn_processed);
return ret;
}
@@ -19610,14 +19672,19 @@ static int do_check_subprogs(struct bpf_verifier_env *env)
{
struct bpf_prog_aux *aux = env->prog->aux;
struct bpf_func_info_aux *sub_aux;
- int i, ret, new_cnt;
+ int context, i, ret, new_cnt;
if (!aux->func_info)
return 0;
- /* exception callback is presumed to be always called */
- if (env->exception_callback_subprog)
- subprog_aux(env, env->exception_callback_subprog)->called = true;
+ /*
+ * Callbacks cannot throw, so the exception callback always runs in the
+ * main program's context. It is presumed to be always called.
+ */
+ if (env->exception_callback_subprog) {
+ sub_aux = subprog_aux(env, env->exception_callback_subprog);
+ sub_aux->called[env->prog->sleepable] = true;
+ }
again:
new_cnt = 0;
@@ -19626,29 +19693,28 @@ again:
continue;
sub_aux = subprog_aux(env, i);
- if (!sub_aux->called || sub_aux->verified)
- continue;
+ for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) {
+ if (!sub_aux->called[context] || sub_aux->verified[context])
+ continue;
- env->insn_idx = env->subprog_info[i].start;
- WARN_ON_ONCE(env->insn_idx == 0);
- ret = do_check_common(env, i);
- if (ret) {
- return ret;
- } else if (env->log.level & BPF_LOG_LEVEL) {
- verbose(env, "Func#%d ('%s') is safe for any args that match its prototype\n",
- i, bpf_subprog_name(env, i));
- }
+ env->insn_idx = env->subprog_info[i].start;
+ WARN_ON_ONCE(env->insn_idx == 0);
+ ret = do_check_common(env, i, context);
+ if (ret)
+ return ret;
+ if (env->log.level & BPF_LOG_LEVEL)
+ verbose(env, "Func#%d ('%s') is safe for any args "
+ "that match its prototype\n",
+ i, bpf_subprog_name(env, i));
- /* We verified new global subprog, it might have called some
- * more global subprogs that we haven't verified yet, so we
- * need to do another pass over subprogs to verify those.
- */
- sub_aux->verified = true;
- new_cnt++;
+ sub_aux->verified[context] = true;
+ new_cnt++;
+ }
}
- /* We can't loop forever as we verify at least one global subprog on
- * each pass.
+ /*
+ * We can't loop forever as each pass verifies at least one new context,
+ * and there are only two contexts per global subprog.
*/
if (new_cnt)
goto again;
@@ -19661,7 +19727,7 @@ static int do_check_main(struct bpf_verifier_env *env)
int ret;
env->insn_idx = 0;
- ret = do_check_common(env, 0);
+ ret = do_check_common(env, 0, env->prog->sleepable);
if (!ret)
env->prog->aux->stack_depth = env->subprog_info[0].stack_depth;
return ret;
@@ -21170,6 +21236,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
ret = bpf_diag_init(env);
if (ret)
goto err_prep;
+ if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) {
+ verbose(env, "invalid bpf_ld_imm64 insn\n");
+ ret = -EINVAL;
+ goto err_prep;
+ }
if (env->signature) {
ret = bpf_prog_calc_tag(env->prog);
if (ret < 0)
@@ -21245,6 +21316,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
if (ret < 0)
goto skip_full_check;
+ /* Apply CO-RE before validating the program's instruction layout. */
+ ret = bpf_check_core_relo(env, attr, uattr);
+ if (ret < 0)
+ goto skip_full_check;
+
/* Discover all subprograms before validating their layout and BTF. */
ret = add_subprogs(env);
if (ret < 0)
@@ -21254,7 +21330,7 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
if (ret < 0)
goto skip_full_check;
- /* Validate BTF against the complete subprogram layout and apply CO-RE. */
+ /* Validate BTF against the complete subprogram layout. */
ret = bpf_check_btf_info(env, attr, uattr);
if (ret < 0)
goto skip_full_check;
diff --git a/kernel/cgroup/cpuset.c b/kernel/cgroup/cpuset.c
index 2538faac9aba..3f52717c1965 100644
--- a/kernel/cgroup/cpuset.c
+++ b/kernel/cgroup/cpuset.c
@@ -1591,10 +1591,11 @@ static int remote_partition_enable(struct cpuset *cs, int new_prs,
* above it or remote partition root underneath it is not allowed.
*/
compute_excpus(cs, tmp->new_cpus);
- WARN_ON_ONCE(cpumask_intersects(tmp->new_cpus, subpartitions_cpus));
if (!cpumask_intersects(tmp->new_cpus, cpu_active_mask) ||
cpumask_subset(top_cpuset.effective_cpus, tmp->new_cpus))
return PERR_INVCPUS;
+ if (cpumask_intersects(tmp->new_cpus, subpartitions_cpus))
+ return PERR_NOCPUS;
if (((new_prs == PRS_ISOLATED) &&
!isolated_cpus_can_update(tmp->new_cpus, NULL)) ||
prstate_housekeeping_conflict(new_prs, tmp->new_cpus))
diff --git a/kernel/cgroup/pids.c b/kernel/cgroup/pids.c
index ecbb839d2acb..78cdc0558d0c 100644
--- a/kernel/cgroup/pids.c
+++ b/kernel/cgroup/pids.c
@@ -253,6 +253,11 @@ static void pids_event(struct pids_cgroup *pids_forking,
}
if (!cgroup_subsys_on_dfl(pids_cgrp_subsys) ||
cgrp_dfl_root.flags & CGRP_ROOT_PIDS_LOCAL_EVENTS) {
+ /*
+ * pids.events reports the local counter on legacy hierarchies
+ * and when pids_localevents is enabled.
+ */
+ cgroup_file_notify(&p->events_file);
cgroup_file_notify(&p->events_local_file);
return;
}
diff --git a/kernel/events/core.c b/kernel/events/core.c
index db7b76d6b68a..634d2ccbab82 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx,
list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) {
cpc = this_cpc(pmu_ctx->pmu);
+ if (cpc->task_epc != pmu_ctx)
+ continue;
+
if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task)
pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in);
}
@@ -3914,7 +3917,7 @@ static void __perf_pmu_sched_task(struct perf_cpu_pmu_context *cpc,
perf_ctx_lock(cpuctx, cpuctx->task_ctx);
perf_pmu_disable(pmu);
- pmu->sched_task(cpc->task_epc, task, sched_in);
+ pmu->sched_task(&cpc->epc, task, sched_in);
perf_pmu_enable(pmu);
perf_ctx_unlock(cpuctx, cpuctx->task_ctx);
@@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev,
struct task_struct *next,
bool sched_in)
{
- struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
struct perf_cpu_pmu_context *cpc, *cpc2;
- /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */
- if (prev == next || cpuctx->task_ctx)
+ if (prev == next)
return;
- list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry)
+ list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) {
+ if (cpc->task_epc)
+ continue;
+
__perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in);
+ }
}
static void perf_event_switch(struct task_struct *task,
@@ -5454,6 +5459,8 @@ attach_task_ctx_data(struct task_struct *task, struct kmem_cache *ctx_cache,
}
if (refcount_inc_not_zero(&old->refcount)) {
+ if (global)
+ old->global = true;
free_perf_ctx_data(cd); /* unused */
return 0;
}
diff --git a/kernel/exit.c b/kernel/exit.c
index 424c44a42a4d..282328d2b4cf 100644
--- a/kernel/exit.c
+++ b/kernel/exit.c
@@ -551,32 +551,6 @@ void mm_update_next_owner(struct mm_struct *mm)
}
#endif /* CONFIG_MEMCG */
-#if defined(CONFIG_SCHED_CACHE) && defined(CONFIG_NUMA_BALANCING)
-/*
- * Subtract the memory footprint of the current task from
- * mm.
- */
-static void exit_mm_sched_cache(struct mm_struct *mm)
-{
- unsigned long fp, sub;
-
- if (!current->total_numa_faults)
- return;
- /*
- * No lock protection due to performance considerations.
- * Make sure mm->sc_stat.footprint does not become
- * negative.
- */
- fp = READ_ONCE(mm->sc_stat.footprint);
- sub = min(fp, current->total_numa_faults);
- WRITE_ONCE(mm->sc_stat.footprint, fp - sub);
-}
-#else
-static inline void exit_mm_sched_cache(struct mm_struct *mm)
-{
-}
-#endif /* CONFIG_SCHED_CACHE CONFIG_NUMA_BALANCING */
-
/*
* Turn us into a lazy TLB process if we
* aren't already..
@@ -589,7 +563,7 @@ static void exit_mm(void)
if (!mm)
return;
- exit_mm_sched_cache(mm);
+ sched_cache_exit_mm(current);
mmap_read_lock(mm);
mmgrab_lazy_tlb(mm);
diff --git a/kernel/fork.c b/kernel/fork.c
index 5ef413368912..10f2d05d816a 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -1599,6 +1599,7 @@ static int copy_mm(u64 clone_flags, struct task_struct *tsk)
tsk->mm = mm;
tsk->active_mm = mm;
+ sched_cache_fork(tsk);
return 0;
}
@@ -2602,6 +2603,7 @@ bad_fork_cleanup_io:
bad_fork_cleanup_namespaces:
exit_nsproxy_namespaces(p);
bad_fork_cleanup_mm:
+ sched_cache_fork_cleanup(p);
if (p->mm) {
mm_clear_owner(p->mm, p);
mmput(p->mm);
diff --git a/kernel/kprobes.c b/kernel/kprobes.c
index 6337da5cab9e..4edd8ca5c657 100644
--- a/kernel/kprobes.c
+++ b/kernel/kprobes.c
@@ -42,6 +42,7 @@
#include <linux/execmem.h>
#include <linux/cleanup.h>
#include <linux/wait.h>
+#include <linux/wait_bit.h>
#include <asm/sections.h>
#include <asm/cacheflush.h>
@@ -526,7 +527,8 @@ enum {
OPTIMIZER_ST_FLUSHING = 2,
};
-static DECLARE_COMPLETION(optimizer_completion);
+/* Bumped at the end of each kprobe_optimizer() pass, under 'kprobe_mutex' */
+static unsigned long optimizer_passes;
#define OPTIMIZE_DELAY 5
@@ -654,9 +656,9 @@ static void kprobe_optimizer(void)
do_free_cleaned_kprobes();
}
- /* Step 5: Kick optimizer again if needed. But if there is a flush requested, */
- if (completion_done(&optimizer_completion))
- complete(&optimizer_completion);
+ /* Step 5: Wake up flushers, and kick optimizer again if needed. */
+ optimizer_passes++;
+ wake_up_var_locked(&optimizer_passes, &kprobe_mutex);
if (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list))
kick_kprobe_optimizer(); /*normal kick*/
@@ -708,7 +710,8 @@ static void wait_for_kprobe_optimizer_locked(void)
lockdep_assert_held(&kprobe_mutex);
while (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) {
- init_completion(&optimizer_completion);
+ unsigned long passes = optimizer_passes;
+
/*
* Set state to OPTIMIZER_ST_FLUSHING and wake up the thread if it's
* idle. If it's already kicked, it will see the state change.
@@ -717,9 +720,12 @@ static void wait_for_kprobe_optimizer_locked(void)
OPTIMIZER_ST_FLUSHING) != OPTIMIZER_ST_FLUSHING)
wake_up(&kprobe_optimizer_wait);
- mutex_unlock(&kprobe_mutex);
- wait_for_completion(&optimizer_completion);
- mutex_lock(&kprobe_mutex);
+ /*
+ * kprobe_optimizer() holds 'kprobe_mutex' for a whole pass, which
+ * this drops while sleeping, so a new count means a full pass ran.
+ */
+ wait_var_event_mutex(&optimizer_passes,
+ optimizer_passes != passes, &kprobe_mutex);
}
}
diff --git a/kernel/power/hibernate.c b/kernel/power/hibernate.c
index d2479c69d71a..c13f68ab7f6e 100644
--- a/kernel/power/hibernate.c
+++ b/kernel/power/hibernate.c
@@ -408,9 +408,18 @@ int hibernation_snapshot(int platform_mode)
if (error)
goto Close;
+ error = dpm_prepare(PMSG_FREEZE);
+ if (error)
+ goto Complete;
+
+ /* Preallocate image memory before freezing kernel threads and shutting down devices. */
+ error = hibernate_preallocate_memory();
+ if (error)
+ goto Complete;
+
error = freeze_kernel_threads();
if (error)
- goto Close;
+ goto Cleanup;
if (hibernation_test(TEST_FREEZER)) {
@@ -422,15 +431,6 @@ int hibernation_snapshot(int platform_mode)
goto Thaw;
}
- error = dpm_prepare(PMSG_FREEZE);
- if (error)
- goto Complete;
-
- /* Preallocate image memory before shutting down devices. */
- error = hibernate_preallocate_memory();
- if (error)
- goto Complete;
-
console_suspend_all();
pm_restrict_gfp_mask();
@@ -464,10 +464,12 @@ int hibernation_snapshot(int platform_mode)
platform_end(platform_mode);
return error;
- Complete:
- dpm_complete(PMSG_RECOVER);
Thaw:
thaw_kernel_threads();
+ Cleanup:
+ swsusp_free();
+ Complete:
+ dpm_complete(PMSG_RECOVER);
goto Close;
}
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 0b846a13c628..1fe40de6ebe3 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5798,7 +5798,7 @@ void sched_tick(void)
curr = rq->curr;
donor = rq->donor;
- psi_account_irqtime(rq, donor, NULL);
+ psi_account_irqtime(rq, curr, NULL);
update_rq_clock(rq);
hw_pressure = arch_scale_hw_pressure(cpu_of(rq));
diff --git a/kernel/sched/ext/cid.c b/kernel/sched/ext/cid.c
index bc4eee5bb4cb..4b08866d75f6 100644
--- a/kernel/sched/ext/cid.c
+++ b/kernel/sched/ext/cid.c
@@ -912,30 +912,36 @@ bool scx_cmask_empty(const struct scx_cmask *m)
/**
* scx_bpf_cid_topo - Copy out per-cid topology info
* @cid: cid to look up
- * @out__uninit: where to copy the topology info; fully written by this call
+ * @out: where to copy the topology info
+ * @out__sz: size of @out, the program's sizeof(struct scx_cid_topo)
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
*
- * Fill @out__uninit with the topology info for @cid. Trigger scx_error() if
- * @cid is out of range. If @cid is valid but in the no-topo section, all fields
- * are set to -1. All fields are also set to -1 when no cid tables have been
- * published yet, which a program may observe while racing the root enable.
+ * Fill @out with the topology info for @cid. Trigger scx_error() if @cid is out
+ * of range. If @cid is valid but in the no-topo section, all fields are set to
+ * -1. All fields are also set to -1 when no cid tables have been published yet,
+ * which a program may observe while racing the root enable.
+ *
+ * The program's struct may be older or newer than the kernel's. The smaller of
+ * @out__sz and the kernel's size is copied and the rest of @out is set to -1.
*/
-__bpf_kfunc void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out__uninit,
+__bpf_kfunc void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out, size_t out__sz,
const struct bpf_prog_aux *aux)
{
+ size_t len = min(out__sz, sizeof(*out));
struct scx_cid_topo *topo;
struct scx_sched *sch;
+ /* the error cases and fields the kernel lacks read as -1 */
+ memset(out, 0xff, out__sz);
+
guard(rcu)();
sch = scx_prog_sched(aux);
topo = rcu_dereference(scx_cid_topo);
- if (unlikely(!sch) || !cid_valid(sch, cid) || unlikely(!topo)) {
- *out__uninit = SCX_CID_TOPO_NEG;
+ if (unlikely(!sch) || !cid_valid(sch, cid) || unlikely(!topo))
return;
- }
- *out__uninit = topo[cid];
+ memcpy(out, &topo[cid], len);
}
__bpf_kfunc_end_defs();
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 3219f0da0fe4..e56c3c95018f 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -447,37 +447,45 @@ static void switch_rq_lock(struct rq *from, struct rq *to)
DEFINE_STATIC_KEY_FALSE(__scx_is_cid_type);
/**
- * scx_call_op_set_cpumask - invoke ops.set_cpumask / ops_cid.set_cmask for @task
+ * scx_fill_cmask_scratch - Build this cpu's arena cmask from @cpumask
+ * @sch: scx_sched whose scratch to fill
+ * @cpumask: cpus to translate into cids
+ *
+ * The scratch lives in BPF-writable arena memory and its header can't be
+ * trusted, so it is rewritten from kernel geometry rather than read. Caller
+ * must hold an rq lock so this cpu is the sole kernel writer for as long as the
+ * returned address is in use.
+ */
+static struct scx_cmask *scx_fill_cmask_scratch(struct scx_sched *sch,
+ const struct cpumask *cpumask)
+{
+ struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch);
+ struct scx_cmask_ref ref;
+
+ scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref);
+ scx_cmask_ref_from_cpumask(&ref, cpumask);
+ return kern_va;
+}
+
+/**
+ * scx_call_op_set_cpumask - Invoke the set_cpumask or set_cmask op for @task
* @sch: scx_sched being invoked
* @rq: rq to update as the currently-locked rq, or NULL
* @task: task whose affinity is changing
* @cpumask: new cpumask
*
- * For cid-form schedulers, translate @cpumask to a cmask via the per-cpu
- * scratch in cid.c and dispatch through the ops_cid union view. Caller
- * must hold @rq's rq lock so this_cpu_ptr is stable across the call.
+ * For cid-form schedulers, translate @cpumask to a cmask in the per-cpu scratch
+ * and dispatch through the ops_cid union view. Caller must hold @rq's rq lock.
*/
static inline void scx_call_op_set_cpumask(struct scx_sched *sch, struct rq *rq,
struct task_struct *task,
const struct cpumask *cpumask)
{
- if (scx_is_cid_type()) {
- struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch);
- struct scx_cmask_ref ref;
-
- /*
- * Build the per-cpu arena cmask from kernel geometry via @ref,
- * never reading its BPF-writable header. set_cmask()'s __arena
- * argument takes the kernel address and the struct_ops
- * trampoline rebases it into BPF's arena pointer form. The rq
- * lock makes this cpu the sole kernel writer.
- */
- scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref);
- scx_cmask_ref_from_cpumask(&ref, cpumask);
- SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task, kern_va);
- } else {
+ if (scx_is_cid_type())
+ SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task,
+ scx_fill_cmask_scratch(sch, cpumask));
+ else
SCX_CALL_OP_TASK(sch, set_cpumask, rq, task, cpumask);
- }
}
enum scx_dsq_iter_flags {
@@ -1499,27 +1507,22 @@ static inline bool task_scx_migrating(struct task_struct *p)
return p->scx.sticky_cpu >= 0;
}
-/*
- * Call ops.dequeue() if the task is in BPF custody and not migrating.
- * Clears %SCX_TASK_IN_CUSTODY when the callback is invoked.
- */
-static void call_task_dequeue(struct scx_sched *sch, struct rq *rq,
- struct task_struct *p, u64 deq_flags)
+/* Must be called under the lock serializing @p's custody transfers. */
+static bool task_leave_custody(struct task_struct *p)
{
if (!(p->scx.flags & SCX_TASK_IN_CUSTODY) || task_scx_migrating(p))
- return;
-
- if (SCX_HAS_OP(sch, dequeue))
- SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags);
+ return false;
p->scx.flags &= ~SCX_TASK_IN_CUSTODY;
+ return true;
}
static void rq_owned_post_enq(struct scx_sched *sch, struct rq *rq,
struct scx_dispatch_q *dsq, struct task_struct *p,
u64 enq_flags)
{
- call_task_dequeue(sch, rq, p, 0);
+ if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0);
/*
* Only local inserts get the wakeup treatment below. Rejects kick the
@@ -1705,20 +1708,28 @@ static void scx_dispatch_enqueue(struct scx_sched *sch, struct rq *rq,
if (is_rq_owned) {
rq_owned_post_enq(sch, rq, dsq, p, enq_flags);
} else {
+ bool call_dequeue = false;
+
/*
* Global and bypass DSQs are terminal - the task leaves the
- * scheduler's custody, so ops.dequeue() fires here. It can run
+ * scheduler's custody, so ops.dequeue() fires. It can run
* without @p's rq lock (finish_dispatch() passes the dispatch
* rq); that's safe because dequeue_task_scx() waits on
* SCX_OPSS_DISPATCHING (see the ops_state note above) and so
* can't race it. A non-terminal DSQ keeps the task in custody.
+ * The custody transfer happens under @dsq->lock so that later
+ * consumers see the flag clear; the callback runs after
+ * @dsq->lock is dropped because it may lock a DSQ itself.
*/
if (dsq->id == SCX_DSQ_GLOBAL || dsq->id == SCX_DSQ_BYPASS)
- call_task_dequeue(sch, rq, p, 0);
+ call_dequeue = task_leave_custody(p);
else
p->scx.flags |= SCX_TASK_IN_CUSTODY;
raw_spin_unlock(&dsq->lock);
+
+ if (call_dequeue && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0);
}
/*
@@ -2141,7 +2152,12 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_
int sticky_cpu = p->scx.sticky_cpu;
u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags;
- if (enq_flags & ENQUEUE_WAKEUP)
+ /*
+ * SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue
+ * returns. Only the core's wakeup path delivers one. The flags stashed
+ * for a remote activation may carry the wakeup bit without it.
+ */
+ if (core_enq_flags & ENQUEUE_WAKEUP)
rq->scx.flags |= SCX_RQ_IN_WAKEUP;
/*
@@ -2210,7 +2226,7 @@ retry:
/*
* A queued task must always be in BPF scheduler's custody. If
* SCX_TASK_IN_CUSTODY is clear, finish_dispatch() on another
- * CPU has already passed call_task_dequeue() (which clears the
+ * CPU has already passed task_leave_custody() (which clears the
* flag), but has not yet written SCX_OPSS_NONE. That final
* store does not require this rq's lock, so retrying with
* cpu_relax() is bounded: we will observe NONE (or DISPATCHING,
@@ -2258,7 +2274,8 @@ retry:
* NONE but the task may still have %SCX_TASK_IN_CUSTODY set until
* it is enqueued on the destination.
*/
- call_task_dequeue(sch, rq, p, deq_flags);
+ if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags);
}
static bool dequeue_task_scx(struct rq *rq, struct task_struct *p, int core_deq_flags)
@@ -2374,14 +2391,10 @@ static void wakeup_preempt_scx(struct rq *rq, struct task_struct *p, int wake_fl
}
void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p,
- u64 enq_flags, struct scx_dispatch_q *src_dsq,
- struct rq *dst_rq)
+ u64 enq_flags, struct rq *dst_rq)
{
struct scx_dispatch_q *dst_dsq = scx_resolve_local_dsq(sch, dst_rq, p, &enq_flags);
- /* @p is on @dst_rq, an rq-owned @src_dsq is covered by the rq lock */
- if (!dsq_is_rq_owned(src_dsq))
- lockdep_assert_held(&src_dsq->lock);
lockdep_assert_rq_held(dst_rq);
WARN_ON_ONCE(p->scx.holding_cpu >= 0);
@@ -2629,8 +2642,8 @@ static struct rq *move_task_between_dsqs(struct scx_sched *sch,
/* @p is going from a non-local DSQ to a local DSQ */
if (src_rq == dst_rq) {
scx_task_unlink_from_dsq(p, src_dsq);
- scx_move_local_task_to_local_dsq(sch, p, enq_flags, src_dsq, dst_rq);
raw_spin_unlock(&src_dsq->lock);
+ scx_move_local_task_to_local_dsq(sch, p, enq_flags, dst_rq);
} else {
raw_spin_unlock(&src_dsq->lock);
move_remote_task_to_local_dsq(sch, p, enq_flags, src_rq, dst_rq);
@@ -2680,8 +2693,8 @@ retry:
if (rq == task_rq) {
scx_task_unlink_from_dsq(p, dsq);
- scx_move_local_task_to_local_dsq(sch, p, enq_flags, dsq, rq);
raw_spin_unlock(&dsq->lock);
+ scx_move_local_task_to_local_dsq(sch, p, enq_flags, rq);
return true;
}
@@ -3629,8 +3642,12 @@ static void set_cpus_allowed_scx(struct task_struct *p,
*
* Fine-grained memory write control is enforced by BPF making the const
* designation pointless. Cast it away when calling the operation.
+ *
+ * The cid form receives the initial mask when the task is enabled and
+ * hears about changes only afterwards, see struct scx_enable_args.
*/
- if (SCX_HAS_OP(sch, set_cpumask))
+ if (SCX_HAS_OP(sch, set_cpumask) &&
+ (!scx_is_cid_type() || scx_get_task_state(p) == SCX_TASK_ENABLED))
scx_call_op_set_cpumask(sch, task_rq(p), p, (struct cpumask *)p->cpus_ptr);
}
@@ -3939,8 +3956,27 @@ static void __scx_enable_task(struct scx_sched *sch, struct task_struct *p)
p->scx.weight = sched_weight_to_cgroup(weight);
- if (SCX_HAS_OP(sch, enable))
- SCX_CALL_OP_TASK(sch, enable, rq, p);
+ if (SCX_HAS_OP(sch, enable)) {
+ if (scx_is_cid_type()) {
+ struct scx_cmask *cmask = scx_fill_cmask_scratch(sch, p->cpus_ptr);
+ struct scx_enable_args args = {
+ .cmask_arena_addr = scx_kaddr_to_arena(sch, cmask),
+ };
+
+ SCX_CALL_CID_OP_TASK(sch, enable, rq, p, &args);
+ } else {
+ SCX_CALL_OP_TASK(sch, enable, rq, p);
+ }
+ }
+
+ /*
+ * The initial mask also goes out through set_cmask() so a scheduler can
+ * track affinity there alone, and before set_weight() so that the mask
+ * is in place when weight-dependent state is derived, see struct
+ * scx_enable_args.
+ */
+ if (scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask))
+ scx_call_op_set_cpumask(sch, rq, p, p->cpus_ptr);
if (SCX_HAS_OP(sch, set_weight))
SCX_CALL_OP_TASK(sch, set_weight, rq, p, p->scx.weight);
@@ -4283,9 +4319,10 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p)
/*
* set_cpus_allowed_scx() is not called while @p is associated with a
- * different scheduler class. Keep the BPF scheduler up-to-date.
+ * different scheduler class. Keep the BPF scheduler up-to-date. The cid
+ * form gets its mask from scx_enable_task().
*/
- if (SCX_HAS_OP(sch, set_cpumask))
+ if (!scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask))
scx_call_op_set_cpumask(sch, rq, p, (struct cpumask *)p->cpus_ptr);
}
@@ -4404,6 +4441,17 @@ static bool local_task_should_reenq(struct rq *rq, struct task_struct *p,
return *reenq_flags & SCX_REENQ_ANY;
}
+/*
+ * The dispatcher stores the final ops_state after dropping the DSQ lock, so @p
+ * can be found on a DSQ while still %SCX_OPSS_DISPATCHING. Reenqueueing @p
+ * before that store lands would have it clobber the new %SCX_OPSS_QUEUED.
+ */
+void scx_reenq_wait_dispatching(struct task_struct *p)
+{
+ if (unlikely(atomic_long_read_acquire(&p->scx.ops_state) == SCX_OPSS_DISPATCHING))
+ wait_ops_state(p, SCX_OPSS_DISPATCHING);
+}
+
static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags)
{
LIST_HEAD(tasks);
@@ -4447,6 +4495,7 @@ static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags)
if (!local_task_should_reenq(rq, p, &reenq_flags, &reason))
continue;
+ scx_reenq_wait_dispatching(p);
scx_dispatch_dequeue(rq, p);
if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK))
@@ -4570,6 +4619,7 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag
}
/* @p is on @dsq, its rq and @dsq are locked */
+ scx_reenq_wait_dispatching(p);
dispatch_dequeue_locked(p, dsq);
raw_spin_unlock(&dsq->lock);
@@ -8356,10 +8406,11 @@ static struct bpf_struct_ops bpf_sched_ext_ops = {
/*
* cid-form cfi stubs. Stubs whose signatures match the cpu-form (param types
* identical, only param names differ across structs) are reused. Some need
- * fresh stubs, set_cmask due to an argument type difference and the sub-sched
- * notifiers because no cpu-form stub exists to reuse.
+ * fresh stubs, set_cmask and enable due to argument differences and the
+ * sub-sched notifiers because no cpu-form stub exists to reuse.
*/
static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx_cmask *cmask__arena) {}
+static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {}
static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
@@ -8380,7 +8431,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.update_idle = sched_ext_ops__update_idle,
.init_task = sched_ext_ops__init_task,
.exit_task = sched_ext_ops__exit_task,
- .enable = sched_ext_ops__enable,
+ .enable = sched_ext_ops_cid__enable,
.disable = sched_ext_ops__disable,
#ifdef CONFIG_EXT_GROUP_SCHED
.cpuctl_init = sched_ext_ops__cgroup_init,
@@ -10403,8 +10454,7 @@ __bpf_kfunc const void *scx_bpf_online_cmask(const struct bpf_prog_aux *aux)
if (unlikely(!online))
return NULL;
- /* BPF rebases by the low 32 bits, like __arena callback args */
- return (void *)((unsigned long)online - sch->arena_kern_base);
+ return (void *)scx_kaddr_to_arena(sch, online);
}
/**
diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h
index ed423bcc26b8..2ff5479334cb 100644
--- a/kernel/sched/ext/inlines.h
+++ b/kernel/sched/ext/inlines.h
@@ -129,8 +129,10 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
* scheduler's ops.dispatch() doesn't yield any tasks.
*/
if (scx_bypass_dsp_enabled(sch) &&
- scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0))
+ scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0)) {
+ __scx_add_event(sch, SCX_EV_SUB_BYPASS_DISPATCH, 1);
return SCX_DSP_LOCAL;
+ }
return SCX_DSP_NONE;
}
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index 3464e0f113c1..1df8f583b0ec 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -250,6 +250,31 @@ struct scx_exit_task_args {
bool cancelled;
};
+/**
+ * struct scx_enable_args - Argument container for cid-form ops.enable()
+ * @cmask_arena_addr: BPF arena address of the cmask of cids the task may run on
+ *
+ * @cmask_arena_addr is the task's affinity as it enters the scheduler.
+ * set_cmask() delivers the same mask right after enable(), before set_weight()
+ * and the first enqueue, then every affinity change afterwards, and is never
+ * called before enable(). A scheduler may therefore track affinity in
+ * set_cmask() alone.
+ *
+ * The kernel builds the mask in the scheduler arena from its own geometry, so
+ * the header is valid regardless of what the scheduler last wrote there. The
+ * memory is per-cpu scratch reused once the callback returns: copy the bits
+ * out, don't keep the address. The set_cmask() argument follows the same rules.
+ *
+ * The address is a plain value rather than a typed pointer because BTF can't
+ * mark a struct member as an arena pointer yet and a pointer member would reach
+ * the program typed as a kernel pointer. Cast it to struct scx_cmask __arena *
+ * before use. Once arena members can be typed, a typed alias will join this
+ * field in an anonymous union at the same offset.
+ */
+struct scx_enable_args {
+ u64 cmask_arena_addr;
+};
+
/* argument container for ops.cgroup_init() */
struct scx_cgroup_init_args {
/* the weight of the cgroup [1..10000] */
@@ -1037,6 +1062,7 @@ struct sched_ext_ops {
* - dispatch -> dispatch (cpu arg is now cid)
* - update_idle -> update_idle (cpu arg is now cid)
* - set_cpumask -> set_cmask (cmask instead of cpumask)
+ * - enable -> enable (takes struct scx_enable_args)
* - cpu_online -> cid_online
* - cpu_offline -> cid_offline
* - dump_cpu -> dump_cid
@@ -1070,7 +1096,7 @@ struct sched_ext_ops_cid {
struct scx_init_task_args *args);
void (*exit_task)(struct task_struct *p,
struct scx_exit_task_args *args);
- void (*enable)(struct task_struct *p);
+ void (*enable)(struct task_struct *p, struct scx_enable_args *args);
void (*disable)(struct task_struct *p);
void (*dump)(struct scx_dump_ctx *ctx);
void (*dump_cid)(struct scx_dump_ctx *ctx, s32 cid, bool idle);
@@ -1533,7 +1559,8 @@ struct scx_sched {
* by BUILD_BUG_ON in scx_init()). The anonymous union lets the kernel
* access either view of the same storage without function-pointer
* casts: use .ops for cpu-form and shared fields, .ops_cid for the
- * cid-renamed callbacks (set_cmask, select_cid, cid_online, ...).
+ * callbacks whose cid-form signature differs (set_cmask, enable,
+ * select_cid, cid_online, ...).
*/
union {
struct sched_ext_ops ops;
@@ -1556,9 +1583,9 @@ struct scx_sched {
uintptr_t arena_kern_base;
/*
- * Per-CPU arena cmask used by scx_call_op_set_cpumask() to hand a cmask
- * to ops_cid.set_cmask(). The kernel writes through the stored kern_va
- * and passes it to the callback's __arena argument.
+ * Per-CPU arena cmask the kernel fills from a task's cpumask and hands
+ * to ops_cid.enable() and ops_cid.set_cmask(). The stored pointers are
+ * the kernel addresses.
*/
struct scx_cmask * __percpu *set_cmask_scratch;
struct scx_cmask *online_cmask;
@@ -1669,6 +1696,19 @@ static inline void *scx_arena_to_kaddr(struct scx_sched *sch, const void *bpf_pt
return (void *)(sch->arena_kern_base + (u32)(uintptr_t)bpf_ptr);
}
+/**
+ * scx_kaddr_to_arena - Translate a kernel arena address to the BPF form
+ * @sch: scheduler whose arena hosts @kaddr
+ * @kaddr: kernel address inside @sch's arena
+ *
+ * __arena callback arguments need no translation. Addresses handed to BPF any
+ * other way, such as struct fields and kfunc return values, go through this.
+ */
+static inline uintptr_t scx_kaddr_to_arena(struct scx_sched *sch, const void *kaddr)
+{
+ return (uintptr_t)kaddr - sch->arena_kern_base;
+}
+
enum scx_wake_flags {
/* expose select WF_* flags as enums */
SCX_WAKE_FORK = WF_FORK,
@@ -2065,8 +2105,7 @@ void scx_dispatch_dequeue(struct rq *rq, struct task_struct *p);
void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags,
int sticky_cpu);
void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p,
- u64 enq_flags, struct scx_dispatch_q *src_dsq,
- struct rq *dst_rq);
+ u64 enq_flags, struct rq *dst_rq);
bool scx_consume_dispatch_q(struct scx_sched *sch, struct rq *rq,
struct scx_dispatch_q *dsq, u64 enq_flags);
bool scx_consume_global_dsq(struct scx_sched *sch, struct rq *rq);
@@ -2078,6 +2117,7 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags);
u64 __scx_bpf_now(struct rq *rq);
void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq,
u64 reenq_flags, struct rq *locked_rq);
+void scx_reenq_wait_dispatching(struct task_struct *p);
int __scx_init_task(struct scx_sched *sch, struct task_struct *p,
struct cgroup *cgrp, bool fork);
void scx_enable_task(struct scx_sched *sch, struct task_struct *p);
@@ -2302,9 +2342,9 @@ do { \
} while (0)
/*
- * Dispatch a task op through the cid-form ops_cid table. Only set_cmask() needs
- * this: it takes an arena cmask address instead of a cpumask, so it cannot be
- * invoked via its cpu-form set_cpumask() slot.
+ * Dispatch a task op through the cid-form ops_cid table, for the ops whose
+ * cid-form signature differs from the cpu-form slot: set_cmask() takes an arena
+ * cmask instead of a cpumask and enable() takes scx_enable_args.
*/
#define SCX_CALL_CID_OP_TASK(sch, op, locked_rq, task, args...) \
__SCX_CALL_OP_TASK(sch, ops_cid, op, locked_rq, task, ##args)
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 34e642a1a403..0472eaf41c7c 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -555,8 +555,8 @@ static void scx_rescue_timerfn(struct timer_list *timer)
scx.dsq_list.node);
scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq);
scx_rescue_admit(rq, p, slice);
- scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS,
- &rq->scx.rescue.dsq, rq);
+ scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
+ SCX_ENQ_IGNORE_CAPS, rq);
if (sched_class_above(&ext_sched_class, rq->curr->sched_class))
resched_curr(rq);
} else if (p->scx.dsq && rq->scx.rescue.budget > 2 * scx_rescue_quantum_ns) {
@@ -572,7 +572,7 @@ static void scx_rescue_timerfn(struct timer_list *timer)
scx_task_unlink_from_dsq(p, &rq->scx.local_dsq);
scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
SCX_ENQ_HEAD | SCX_ENQ_PREEMPT | SCX_ENQ_IGNORE_CAPS,
- &rq->scx.local_dsq, rq);
+ rq);
}
out_arm:
scx_rescue_timer_arm(rq);
@@ -596,8 +596,8 @@ void scx_rescue_flush(struct rq *rq)
/* and flush out all pending ones */
list_for_each_entry_safe(p, n, &rq->scx.rescue.dsq.list, scx.dsq_list.node) {
scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq);
- scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS,
- &rq->scx.rescue.dsq, rq);
+ scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
+ SCX_ENQ_IGNORE_CAPS, rq);
}
timer_delete(&rq->scx.rescue.timer);
@@ -801,6 +801,7 @@ void scx_reenq_reject(struct rq *rq)
if (WARN_ON_ONCE(p->migration_pending))
continue;
+ scx_reenq_wait_dispatching(p);
scx_dispatch_dequeue(rq, p);
if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK))
diff --git a/kernel/sched/ext/types.h b/kernel/sched/ext/types.h
index 943d8d429a2c..139176cf9fc6 100644
--- a/kernel/sched/ext/types.h
+++ b/kernel/sched/ext/types.h
@@ -70,6 +70,10 @@ enum scx_consts {
* smaller shards if the LLC exceeds the target size. No-topo cids are packed
* into their own max-sized shards.
*
+ * New fields are appended, never inserted: scx_bpf_cid_topo() copies this
+ * struct out sized by the program's own layout, and an older program's copy
+ * must stay a prefix of the kernel's.
+ *
* @core_cid: first cid of this cid's core (smt-sibling group)
* @core_idx: global index of that core, in [0, nr_cores_at_init)
* @llc_cid: first cid of this cid's LLC
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 7455a83a6a99..57360f5cdde4 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -1478,7 +1478,7 @@ static inline int get_sched_cache_scale(int mul)
return (1 + (tol - 1) * mul);
}
-static bool exceed_llc_capacity(struct mm_struct *mm, int cpu)
+static bool exceed_llc_capacity(struct sched_cache_group *grp, int cpu)
{
#ifdef CONFIG_NUMA_BALANCING
unsigned long llc, footprint;
@@ -1497,7 +1497,7 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu)
* excluded.
*/
llc = sd->llc_bytes;
- footprint = READ_ONCE(mm->sc_stat.footprint);
+ footprint = READ_ONCE(grp->footprint);
/*
* Scale the LLC size by 256*llc_aggr_tolerance
@@ -1526,7 +1526,7 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu)
return false;
}
-static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p,
+static bool invalid_llc_nr(struct sched_cache_group *grp, struct task_struct *p,
int cpu)
{
int scale;
@@ -1542,10 +1542,32 @@ static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p,
if (scale == INT_MAX)
return false;
- return !fits_capacity((mm->sc_stat.nr_running_avg * cpu_smt_num_threads),
+ return !fits_capacity((READ_ONCE(grp->nr_running_avg) * cpu_smt_num_threads),
(scale * per_cpu(sd_llc_size, cpu)));
}
+/*
+ * A task counts in nr_pref_llc_running while it is queued on its preferred
+ * LLC (pref_llc_queued) and runnable (!sched_delayed), keeping the counter in
+ * the runnable domain so alb_break_llc() can compare it with h_nr_runnable.
+ */
+static bool task_pref_llc_runnable(struct task_struct *p)
+{
+ return p->pref_llc_queued && !p->se.sched_delayed;
+}
+
+static void pref_llc_running_inc(struct rq *rq, struct task_struct *p)
+{
+ if (task_pref_llc_runnable(p))
+ rq->nr_pref_llc_running++;
+}
+
+static void pref_llc_running_dec(struct rq *rq, struct task_struct *p)
+{
+ if (task_pref_llc_runnable(p))
+ rq->nr_pref_llc_running--;
+}
+
static void account_llc_enqueue(struct rq *rq, struct task_struct *p)
{
int pref_llc, pref_llc_queued;
@@ -1557,7 +1579,6 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p)
pref_llc_queued = (pref_llc == task_llc(p));
rq->nr_llc_running++;
- rq->nr_pref_llc_running += pref_llc_queued;
/*
* Record whether p is enqueued on its preferred
@@ -1575,6 +1596,9 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p)
*/
p->pref_llc_queued = pref_llc_queued;
+ /* Skipped while delayed; clear_delayed() adds it back on wake. */
+ pref_llc_running_inc(rq, p);
+
sd = rcu_dereference_all(rq->sd);
if (sd && (unsigned int)pref_llc < sd->llc_max)
sd->llc_counts[pref_llc]++;
@@ -1591,7 +1615,12 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p)
rq->nr_llc_running--;
if (p->pref_llc_queued) {
- rq->nr_pref_llc_running--;
+ /*
+ * Skipped if still delayed (set_delayed() already removed it);
+ * clearing pref_llc_queued below also stops clear_delayed()
+ * from re-adding it.
+ */
+ pref_llc_running_dec(rq, p);
/*
* Update the status in case
* other logic might query
@@ -1619,12 +1648,20 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p)
}
}
-void mm_init_sched(struct mm_struct *mm,
- struct sched_cache_time __percpu *_pcpu_sched)
+int mm_init_sched(struct mm_struct *mm,
+ struct sched_cache_time __percpu *_pcpu_sched)
{
+ struct sched_cache_group *grp;
unsigned long epoch = 0;
int i;
+ grp = kzalloc_obj(*grp);
+ if (!grp) {
+ free_percpu(_pcpu_sched);
+ mm->sched_cache_grp = NULL;
+ return -ENOMEM;
+ }
+
for_each_possible_cpu(i) {
struct sched_cache_time *pcpu_sched = per_cpu_ptr(_pcpu_sched, i);
struct rq *rq = cpu_rq(i);
@@ -1635,18 +1672,141 @@ void mm_init_sched(struct mm_struct *mm,
epoch = rq->cpu_epoch;
}
- raw_spin_lock_init(&mm->sc_stat.lock);
- mm->sc_stat.epoch = epoch;
- mm->sc_stat.cpu = -1;
- mm->sc_stat.next_scan = jiffies;
- mm->sc_stat.nr_running_avg = 0;
- mm->sc_stat.footprint = 0;
+ raw_spin_lock_init(&grp->lock);
+ grp->epoch = epoch;
+ grp->cpu = -1;
+ grp->next_scan = jiffies;
+ grp->nr_running_avg = 0;
+ grp->footprint = 0;
+ refcount_set(&grp->refcnt, 1);
/*
- * The update to mm->sc_stat should not be reordered
- * before initialization to mm's other fields, in case
+ * The update to grp->pcpu_sched should not be reordered
+ * before initialization to grp's other fields, in case
* the readers may get invalid mm_sched_epoch, etc.
*/
- smp_store_release(&mm->sc_stat.pcpu_sched, _pcpu_sched);
+ smp_store_release(&grp->pcpu_sched, _pcpu_sched);
+ /*
+ * Publish the group last. Not every reader qualifies it by
+ * grp->pcpu_sched - can_migrate_llc_task() only checks that the
+ * pointer is non-NULL before reading grp->footprint and
+ * grp->nr_running_avg - so a reachable group must already be
+ * fully initialized.
+ */
+ smp_store_release(&mm->sched_cache_grp, grp);
+ return 0;
+}
+
+static void sched_cache_group_free_rcu(struct rcu_head *rcu)
+{
+ struct sched_cache_group *grp =
+ container_of(rcu, struct sched_cache_group, rcu);
+
+ free_percpu(grp->pcpu_sched);
+ kfree(grp);
+}
+
+static void sched_cache_group_put(struct sched_cache_group *grp)
+{
+ if (!grp || !refcount_dec_and_test(&grp->refcnt))
+ return;
+
+ call_rcu(&grp->rcu, sched_cache_group_free_rcu);
+}
+
+DEFINE_FREE(sched_cache_group_put, struct sched_cache_group *,
+ sched_cache_group_put(_T));
+
+#define rcu_deref_sched_cache_grp(tsk) \
+ rcu_dereference_check((tsk)->sched_cache_grp, (tsk) == current)
+
+static struct sched_cache_group *sched_cache_replace_grp(struct task_struct *p,
+ struct sched_cache_group *new)
+{
+ struct sched_cache_group *old;
+
+ old = rcu_deref_sched_cache_grp(p);
+ rcu_assign_pointer(p->sched_cache_grp, new);
+
+ return old;
+}
+
+struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp)
+{
+ /*
+ * refcount_inc_not_zero() is the acquire primitive for lockless
+ * (RCU) lookups; plain refcount_inc() would scribble the count if
+ * it already reached zero. Return NULL in that case.
+ */
+ if (grp && !refcount_inc_not_zero(&grp->refcnt))
+ grp = NULL;
+
+ return grp;
+}
+
+struct sched_cache_group *task_cache_group_get(struct task_struct *p)
+{
+ guard(rcu)();
+ return sched_cache_group_get(rcu_dereference(p->sched_cache_grp));
+}
+
+void sched_cache_fork(struct task_struct *p)
+{
+ /*
+ * The child takes its own reference on the mm's cache group, separate
+ * from the reference held by the mm. @p is not yet visible to readers,
+ * so a plain initializing store is enough.
+ */
+ RCU_INIT_POINTER(p->sched_cache_grp,
+ sched_cache_group_get(p->mm->sched_cache_grp));
+}
+
+void sched_cache_fork_cleanup(struct task_struct *p)
+{
+ /*
+ * A fork that fails after sched_cache_fork() never reaches exit_mm(),
+ * so drop the reference here. @p never became visible, so there are no
+ * concurrent readers and the reference we hold keeps the group alive.
+ */
+ sched_cache_group_put(rcu_access_pointer(p->sched_cache_grp));
+ RCU_INIT_POINTER(p->sched_cache_grp, NULL);
+}
+
+void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm)
+{
+ struct sched_cache_group *old;
+
+ /*
+ * Acquire the new reference before publishing the pointer, then drop
+ * the old one. @p is current and the only writer of its own pointer.
+ */
+ old = sched_cache_replace_grp(p, sched_cache_group_get(mm->sched_cache_grp));
+ sched_cache_group_put(old);
+}
+
+void sched_cache_exit_mm(struct task_struct *p)
+{
+ struct sched_cache_group *grp = sched_cache_replace_grp(p, NULL);
+
+#ifdef CONFIG_NUMA_BALANCING
+ /*
+ * Subtract this task's footprint from the group before dropping the
+ * reference, so the group footprint converges as its threads exit.
+ * Unlocked for performance; clamp to avoid underflow.
+ */
+ if (grp && p->total_numa_faults) {
+ unsigned long fp = READ_ONCE(grp->footprint);
+ unsigned long sub = min(fp, p->total_numa_faults);
+
+ WRITE_ONCE(grp->footprint, fp - sub);
+ }
+#endif
+ sched_cache_group_put(grp);
+}
+
+void mm_destroy_sched(struct mm_struct *mm)
+{
+ sched_cache_group_put(mm->sched_cache_grp);
+ mm->sched_cache_grp = NULL;
}
/* because why would C be fully specified */
@@ -1697,14 +1857,14 @@ static unsigned long fraction_mm_sched(struct rq *rq,
return div64_u64(NICE_0_LOAD * pcpu_sched->runtime, rq->cpu_runtime + 1);
}
-static int get_pref_llc(struct task_struct *p, struct mm_struct *mm)
+static int get_pref_llc(struct task_struct *p, struct sched_cache_group *grp)
{
int mm_sched_llc = -1, mm_sched_cpu;
- if (!mm)
+ if (!grp)
return -1;
- mm_sched_cpu = READ_ONCE(mm->sc_stat.cpu);
+ mm_sched_cpu = READ_ONCE(grp->cpu);
if (mm_sched_cpu != -1) {
mm_sched_llc = llc_id(mm_sched_cpu);
@@ -1734,8 +1894,8 @@ static unsigned int task_running_on_cpu(int cpu, struct task_struct *p);
static inline
void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec)
{
+ struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp);
struct sched_cache_time *pcpu_sched;
- struct mm_struct *mm = p->mm;
int mm_sched_llc = -1;
unsigned long epoch;
@@ -1746,12 +1906,18 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec)
return;
/*
* init_task, kthreads and user thread created
- * by user_mode_thread() don't have mm.
+ * by user_mode_thread() don't have a cache group.
+ * In theory a kernel thread does not have any valid
+ * cache group, because sched_cache_fork() is not
+ * invoked for a kernel thread - !grp should gate the
+ * kernel thread. Use the PF_KTHREAD check explicitly
+ * here for safety reasons, to guard against future
+ * modifications and to pair with task_tick_cache().
*/
- if (!mm || !mm->sc_stat.pcpu_sched)
+ if (p->flags & PF_KTHREAD || !grp || !grp->pcpu_sched)
return;
- pcpu_sched = per_cpu_ptr(mm->sc_stat.pcpu_sched, cpu_of(rq));
+ pcpu_sched = per_cpu_ptr(grp->pcpu_sched, cpu_of(rq));
scoped_guard (raw_spinlock, &rq->cpu_epoch_lock) {
__update_mm_sched(rq, pcpu_sched);
@@ -1764,14 +1930,14 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec)
* If this process hasn't hit task_cache_work() for a while invalidate
* its preferred state.
*/
- if ((long)(epoch - READ_ONCE(mm->sc_stat.epoch)) > llc_epoch_affinity_timeout ||
- invalid_llc_nr(mm, p, cpu_of(rq)) ||
- exceed_llc_capacity(mm, cpu_of(rq))) {
- if (READ_ONCE(mm->sc_stat.cpu) != -1)
- WRITE_ONCE(mm->sc_stat.cpu, -1);
+ if ((long)(epoch - READ_ONCE(grp->epoch)) > llc_epoch_affinity_timeout ||
+ invalid_llc_nr(grp, p, cpu_of(rq)) ||
+ exceed_llc_capacity(grp, cpu_of(rq))) {
+ if (READ_ONCE(grp->cpu) != -1)
+ WRITE_ONCE(grp->cpu, -1);
}
- mm_sched_llc = get_pref_llc(p, mm);
+ mm_sched_llc = get_pref_llc(p, grp);
/* task not on rq accounted later in account_entity_enqueue() */
if (task_running_on_cpu(rq->cpu, p) &&
@@ -1784,31 +1950,32 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec)
static void task_tick_cache(struct rq *rq, struct task_struct *p)
{
+ struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp);
struct callback_head *work = &p->cache_work;
- struct mm_struct *mm = p->mm;
unsigned long epoch;
if (!sched_cache_enabled())
return;
- if (!mm || p->flags & PF_KTHREAD ||
- !mm->sc_stat.pcpu_sched)
+ if (!grp || p->flags & PF_KTHREAD ||
+ !grp->pcpu_sched)
return;
epoch = rq->cpu_epoch;
/* avoid moving backwards */
- if (time_after_eq(mm->sc_stat.epoch, epoch))
+ if (time_after_eq(grp->epoch, epoch))
return;
- guard(raw_spinlock)(&mm->sc_stat.lock);
+ guard(raw_spinlock)(&grp->lock);
if (work->next == work) {
task_work_add(p, work, TWA_RESUME);
- WRITE_ONCE(mm->sc_stat.epoch, epoch);
+ WRITE_ONCE(grp->epoch, epoch);
}
}
-static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p)
+static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p,
+ struct sched_cache_group *grp)
{
#ifdef CONFIG_NUMA_BALANCING
int cpu, curr_cpu, nid, pref_nid;
@@ -1816,7 +1983,7 @@ static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p)
if (!static_branch_likely(&sched_numa_balancing))
goto out;
- cpu = READ_ONCE(p->mm->sc_stat.cpu);
+ cpu = READ_ONCE(grp->cpu);
if (cpu != -1)
nid = cpu_to_node(cpu);
curr_cpu = task_cpu(p);
@@ -1873,13 +2040,13 @@ static inline void update_avg_scale(u64 *avg, u64 sample)
static void task_cache_work(struct callback_head *work)
{
+ struct sched_cache_group *grp __free(sched_cache_group_put) = NULL;
+ cpumask_var_t cpus __free(free_cpumask_var) = CPUMASK_VAR_NULL;
int cpu, m_a_cpu = -1, nr_running = 0, curr_cpu;
unsigned long next_scan, now = jiffies;
struct task_struct *p = current, *cur;
unsigned long curr_m_a_occ = 0;
- struct mm_struct *mm = p->mm;
unsigned long m_a_occ = 0;
- cpumask_var_t cpus;
WARN_ON_ONCE(work != &p->cache_work);
@@ -1888,21 +2055,30 @@ static void task_cache_work(struct callback_head *work)
if (p->flags & PF_EXITING)
return;
- next_scan = READ_ONCE(mm->sc_stat.next_scan);
+ /*
+ * A reference makes sure grp is not released by others. The rcu
+ * lock can not be held till after zalloc_cpumask_var() below,
+ * because the latter might sleep.
+ */
+ grp = task_cache_group_get(p);
+ if (!grp)
+ return;
+
+ next_scan = READ_ONCE(grp->next_scan);
if (time_before(now, next_scan))
return;
/* only 1 thread is allowed to scan */
- if (!try_cmpxchg(&mm->sc_stat.next_scan, &next_scan,
+ if (!try_cmpxchg(&grp->next_scan, &next_scan,
now + max_t(unsigned long,
READ_ONCE(llc_epoch_period), 1)))
return;
curr_cpu = task_cpu(p);
- if (invalid_llc_nr(mm, p, curr_cpu) ||
- exceed_llc_capacity(mm, curr_cpu)) {
- if (READ_ONCE(mm->sc_stat.cpu) != -1)
- WRITE_ONCE(mm->sc_stat.cpu, -1);
+ if (invalid_llc_nr(grp, p, curr_cpu) ||
+ exceed_llc_capacity(grp, curr_cpu)) {
+ if (READ_ONCE(grp->cpu) != -1)
+ WRITE_ONCE(grp->cpu, -1);
return;
}
@@ -1913,7 +2089,7 @@ static void task_cache_work(struct callback_head *work)
scoped_guard (cpus_read_lock) {
guard(rcu)();
- get_scan_cpumasks(cpus, p);
+ get_scan_cpumasks(cpus, p, grp);
for_each_cpu(cpu, cpus) {
/* XXX sched_cluster_active */
@@ -1926,16 +2102,20 @@ static void task_cache_work(struct callback_head *work)
for_each_cpu(i, sched_domain_span(sd)) {
occ = fraction_mm_sched(cpu_rq(i),
- per_cpu_ptr(mm->sc_stat.pcpu_sched, i));
+ per_cpu_ptr(grp->pcpu_sched, i));
a_occ += occ;
if (occ > m_occ) {
m_occ = occ;
m_cpu = i;
}
+ /*
+ * rcu_access_pointer() is used because the
+ * pointer is only compared, never dereferenced.
+ */
cur = rcu_dereference_all(cpu_rq(i)->curr);
if (cur && !(cur->flags & (PF_EXITING | PF_KTHREAD)) &&
- cur->mm == mm)
+ rcu_access_pointer(cur->sched_cache_grp) == grp)
nr_running++;
}
@@ -1959,7 +2139,7 @@ static void task_cache_work(struct callback_head *work)
m_a_cpu = m_cpu;
}
- if (llc_id(cpu) == llc_id(READ_ONCE(mm->sc_stat.cpu)))
+ if (llc_id(cpu) == llc_id(READ_ONCE(grp->cpu)))
curr_m_a_occ = a_occ;
cpumask_andnot(cpus, cpus, sched_domain_span(sd));
@@ -1968,7 +2148,7 @@ static void task_cache_work(struct callback_head *work)
if (m_a_occ > (2 * curr_m_a_occ)) {
/*
- * Avoid switching sc_stat.cpu too fast.
+ * Avoid switching sched_cache_grp->cpu too fast.
* The reason to choose 2X is because:
* 1. It is better to keep the preferred LLC stable,
* rather than changing it frequently and cause migrations
@@ -1977,11 +2157,10 @@ static void task_cache_work(struct callback_head *work)
* 3. 2X is chosen based on test results, as it delivers
* the optimal performance gain so far.
*/
- WRITE_ONCE(mm->sc_stat.cpu, m_a_cpu);
+ WRITE_ONCE(grp->cpu, m_a_cpu);
}
- update_avg_scale(&mm->sc_stat.nr_running_avg, nr_running);
- free_cpumask_var(cpus);
+ update_avg_scale(&grp->nr_running_avg, nr_running);
}
void init_sched_mm(struct task_struct *p)
@@ -1991,10 +2170,18 @@ void init_sched_mm(struct task_struct *p)
init_task_work(work, task_cache_work);
work->next = work;
/*
+ * dup_task_struct() copies the parent's task_struct, including its
+ * sched_cache_grp, for which the child holds no reference. Clear it
+ * here - before copy_mm() runs - so the child never carries a
+ * borrowed pointer that the fork error path would put.
+ */
+ RCU_INIT_POINTER(p->sched_cache_grp, NULL);
+ /*
* Reset new task's preference to avoid
* polluting account_llc_enqueue().
*/
p->preferred_llc = -1;
+ p->pref_llc_queued = 0;
}
#else /* CONFIG_SCHED_CACHE */
@@ -2016,6 +2203,10 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) {}
static void account_llc_dequeue(struct rq *rq, struct task_struct *p) {}
+static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) {}
+
+static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) {}
+
#endif /* CONFIG_SCHED_CACHE */
/*
@@ -3692,6 +3883,7 @@ static int preferred_group_nid(struct task_struct *p, int nid)
static void task_numa_placement(struct task_struct *p)
__context_unsafe(/* conditional locking */)
{
+ struct sched_cache_group __maybe_unused *grp;
int seq, nid, max_nid = NUMA_NO_NODE;
unsigned long max_faults = 0;
unsigned long fault_types[2] = { 0, 0 };
@@ -3784,19 +3976,24 @@ static void task_numa_placement(struct task_struct *p)
* heuristic and occasional lost updates are tolerable.
*
* If a task exits, its corresponding footprint must
- * be subtracted from the mm->sc_stat.footprint, otherwise
- * the mm->sc_stat.footprint will not converge:
- * the exiting thread's footprint remains unchanged/undecayed
- * in mm->sc_stat.footprint. See exit_mm().
+ * be subtracted from p->sched_cache_grp->footprint,
+ * otherwise the footprint will not converge: the
+ * exiting thread's footprint remains unchanged/undecayed.
+ * See exit_mm().
*
* Lost updates and unsynchronized subtraction
* in exit_mm() can cause footprint + diff to
* go negative. Clamp to zero to prevent the
* unsigned footprint from wrapping.
*/
- new_fp = (long)READ_ONCE(p->mm->sc_stat.footprint) + diff;
- WRITE_ONCE(p->mm->sc_stat.footprint,
- max(new_fp, 0L));
+ scoped_guard(rcu) {
+ grp = rcu_dereference(p->sched_cache_grp);
+
+ if (grp) {
+ new_fp = (long)READ_ONCE(grp->footprint) + diff;
+ WRITE_ONCE(grp->footprint, max(new_fp, 0L));
+ }
+ }
#endif
}
@@ -6390,15 +6587,27 @@ static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq);
static void set_delayed(struct sched_entity *se)
{
- se->sched_delayed = 1;
-
/*
* Delayed se of cfs_rq have no tasks queued on them.
* Do not adjust h_nr_runnable since __dequeue_task()
* will account it for blocked tasks.
+ *
+ * This check can be removed because when flat pick
+ * patches get merged as only task can get delayed,
+ * same for clear_delayed().
*/
- if (!entity_is_task(se))
+ if (!entity_is_task(se)) {
+ se->sched_delayed = 1;
return;
+ }
+
+ /*
+ * Drop a task leaving the runnable set.
+ * Needs to be called before sched_delayed is set.
+ * clear_delayed() mirrors this after clearing the flag.
+ */
+ pref_llc_running_dec(rq_of(cfs_rq_of(se)), task_of(se));
+ se->sched_delayed = 1;
for_each_sched_entity(se) {
struct cfs_rq *cfs_rq = cfs_rq_of(se);
@@ -6420,6 +6629,13 @@ static void clear_delayed(struct sched_entity *se)
if (!entity_is_task(se))
return;
+ /*
+ * Re-add on wake, after sched_delayed is cleared. On a final delayed
+ * dequeue account_llc_dequeue() already cleared pref_llc_queued, so
+ * this does nothing.
+ */
+ pref_llc_running_inc(rq_of(cfs_rq_of(se)), task_of(se));
+
for_each_sched_entity(se) {
struct cfs_rq *cfs_rq = cfs_rq_of(se);
@@ -10395,6 +10611,7 @@ enum migration_type {
#define LBF_SOME_PINNED 0x08
#define LBF_ACTIVE_LB 0x10
#define LBF_LLC_PINNED 0x20
+#define LBF_ACTIVE_LB_LLC 0x40
struct lb_env {
struct sched_domain *sd;
@@ -10724,7 +10941,7 @@ static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct
static enum llc_mig can_migrate_llc_task(struct lb_env *env,
struct task_struct *p)
{
- struct mm_struct *mm;
+ struct sched_cache_group *grp;
bool to_pref;
int cpu, src_cpu, dst_cpu;
@@ -10733,19 +10950,19 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env,
src_cpu = env->src_cpu;
dst_cpu = env->dst_cpu;
- mm = p->mm;
- if (!mm)
+ grp = rcu_dereference_all(p->sched_cache_grp);
+ if (!grp)
return mig_unrestricted;
- cpu = READ_ONCE(mm->sc_stat.cpu);
+ cpu = READ_ONCE(grp->cpu);
if (cpu < 0 || cpus_share_cache(src_cpu, dst_cpu))
return mig_unrestricted;
/* skip cache aware load balance for too many threads */
- if (invalid_llc_nr(mm, p, dst_cpu) ||
- exceed_llc_capacity(mm, dst_cpu)) {
- if (READ_ONCE(mm->sc_stat.cpu) != -1)
- WRITE_ONCE(mm->sc_stat.cpu, -1);
+ if (invalid_llc_nr(grp, p, dst_cpu) ||
+ exceed_llc_capacity(grp, dst_cpu)) {
+ if (READ_ONCE(grp->cpu) != -1)
+ WRITE_ONCE(grp->cpu, -1);
return mig_unrestricted;
}
@@ -10814,6 +11031,21 @@ alb_break_llc(struct lb_env *env)
}
/*
+ * Returns true if p's preferred LLC does not match the destination CPU
+ * under migrate_llc_task semantics. Passive LB passes migrate_llc_task
+ * in env->migration_type, while active LB carries LBF_ACTIVE_LB_LLC in
+ * env->flags to avoid overwriting env->migration_type.
+ */
+static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+ return sched_cache_enabled() &&
+ (env->migration_type == migrate_llc_task ||
+ env->flags & LBF_ACTIVE_LB_LLC) &&
+ READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu);
+}
+
+/*
* Check if migrating task p from env->src_cpu to
* env->dst_cpu breaks LLC localiy.
*/
@@ -10841,8 +11073,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
* run on env->dst_cpu, skip the tasks do not prefer
* env->dst_cpu, and find the one that prefers.
*/
- if (env->migration_type == migrate_llc_task &&
- READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
+ if (migrate_llc_task_wrong_dst(p, env))
return true;
if (can_migrate_llc_task(env, p) != mig_forbid)
@@ -10865,6 +11096,12 @@ alb_break_llc(struct lb_env *env)
}
static inline bool
+migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
+{
+ return false;
+}
+
+static inline bool
migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
{
return false;
@@ -10963,7 +11200,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env)
* 4) too many balance attempts have failed.
*/
if (env->flags & LBF_ACTIVE_LB)
- return 1;
+ return !migrate_llc_task_wrong_dst(p, env);
degrades = migrate_degrades_locality(p, env);
if (!degrades) {
@@ -13362,6 +13599,20 @@ static int need_active_balance(struct lb_env *env)
}
static int active_load_balance_cpu_stop(void *data);
+static int active_load_balance_llc_cpu_stop(void *data);
+
+/*
+ * migration_type is checked elsewhere to decide migration policy, so
+ * it shouldn't be repurposed just to flag an LLC-directed active
+ * balance across the stopper. Pick the callback here instead.
+ */
+static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env)
+{
+ if (env->migration_type == migrate_llc_task)
+ return active_load_balance_llc_cpu_stop;
+
+ return active_load_balance_cpu_stop;
+}
static int should_we_balance(struct lb_env *env)
{
@@ -13707,7 +13958,7 @@ more_balance:
}
if (active_balance) {
stop_one_cpu_nowait(cpu_of(busiest),
- active_load_balance_cpu_stop, busiest,
+ alb_stop_fn(&env), busiest,
&busiest->active_balance_work);
}
preempt_enable();
@@ -13812,7 +14063,7 @@ update_next_balance(struct sched_domain *sd, unsigned long *next_balance)
* least 1 task to be running on each physical CPU where possible, and
* avoids physical / logical imbalances.
*/
-static int active_load_balance_cpu_stop(void *data)
+static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags)
{
struct rq *busiest_rq = data;
int busiest_cpu = cpu_of(busiest_rq);
@@ -13862,7 +14113,7 @@ static int active_load_balance_cpu_stop(void *data)
.src_cpu = busiest_rq->cpu,
.src_rq = busiest_rq,
.idle = CPU_IDLE,
- .flags = LBF_ACTIVE_LB,
+ .flags = LBF_ACTIVE_LB | lb_flags,
};
schedstat_inc(sd->alb_count);
@@ -13890,6 +14141,16 @@ out_unlock:
return 0;
}
+static int active_load_balance_cpu_stop(void *data)
+{
+ return __active_load_balance_cpu_stop(data, 0);
+}
+
+static int active_load_balance_llc_cpu_stop(void *data)
+{
+ return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC);
+}
+
/*
* Scale the max sched_balance_rq interval with the number of CPUs in the system.
* This trades load-balance latency on larger machines for less cross talk.
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 0248227d983a..3dab0253976f 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -985,8 +985,8 @@ void sched_cache_active_set(void)
}
/*
- * Update the bottom sched_domain's llc_bytes for @cpu and all its
- * LLC siblings. Called from cacheinfo_cpu_online() or
+ * Update the bottom sched_domain's llc_bytes for @cpus sharing a physical
+ * LLC. Called from cacheinfo_cpu_online() or
* cacheinfo_cpu_pre_down() with cpu hotplug lock held.
*
* Note: get_effective_llc_bytes() returns 0 on PowerPC.
@@ -996,17 +996,13 @@ void sched_cache_active_set(void)
* and does not populates the per-CPU struct cpu_cacheinfo array
* that get_cpu_cacheinfo_llc() reads.
*/
-void sched_update_llc_bytes(unsigned int cpu)
+void sched_update_llc_bytes(const struct cpumask *cpus)
{
struct sched_domain *sd, *sdp;
unsigned int i;
sched_domains_mutex_lock();
- sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, cpu));
- if (!sdp)
- goto unlock;
-
/*
* ci->shared_cpu_map is built incrementally as CPUs come
* online, so the first CPU in an LLC initially sees
@@ -1014,14 +1010,22 @@ void sched_update_llc_bytes(unsigned int cpu)
* get_effective_llc_bytes(). Re-evaluating every LLC
* sibling on each online event corrects this once the full
* shared_cpu_map is known.
+ *
+ * The departing CPU's domains have already been detached when
+ * cacheinfo removes it. Use the surviving cache siblings instead.
+ * They may belong to different cpuset partitions, so use each CPU's
+ * own LLC domain to scale its share of the physical cache.
*/
- for_each_cpu(i, sched_domain_span(sdp)) {
+ for_each_cpu(i, cpus) {
+ sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, i));
+ if (!sdp)
+ continue;
+
sd = rcu_dereference_sched_domain(cpu_rq(i)->sd);
if (sd)
sd->llc_bytes = get_effective_llc_bytes(i, sdp);
}
-unlock:
sched_domains_mutex_unlock();
}
diff --git a/kernel/trace/fprobe.c b/kernel/trace/fprobe.c
index 1e9b00997ff2..9f2d98181779 100644
--- a/kernel/trace/fprobe.c
+++ b/kernel/trace/fprobe.c
@@ -171,6 +171,11 @@ static inline bool write_fprobe_header(unsigned long *stack,
static inline void read_fprobe_header(unsigned long *stack,
struct fprobe **fp, unsigned int *size_words)
{
+ if (!*stack) {
+ *fp = NULL;
+ *size_words = 0;
+ return;
+ }
*fp = arch_decode_fprobe_header_fp(*stack);
*size_words = arch_decode_fprobe_header_size(*stack);
}
@@ -203,6 +208,12 @@ static inline void read_fprobe_header(unsigned long *stack,
{
struct __fprobe_header *fph = (struct __fprobe_header *)stack;
+ if (!*stack) {
+ *fp = NULL;
+ *size_words = 0;
+ return;
+ }
+
*fp = fph->fp;
*size_words = fph->size_words;
}
@@ -635,6 +646,10 @@ static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops
}
}
+ /* Terminate the list, fgraph_reserve_data() does not clear it. */
+ if (used && used < reserved_words)
+ fgraph_data[used] = 0;
+
/* If any exit_handler is set, data must be used. */
return used != 0;
}
diff --git a/kernel/workqueue.c b/kernel/workqueue.c
index 1ae3732a2c51..959525393739 100644
--- a/kernel/workqueue.c
+++ b/kernel/workqueue.c
@@ -3918,7 +3918,7 @@ static void check_flush_dependency(struct workqueue_struct *target_wq,
WARN_ONCE(current->flags & PF_MEMALLOC,
"workqueue: PF_MEMALLOC task %d(%s) is flushing !WQ_MEM_RECLAIM %s:%ps",
current->pid, current->comm, target_wq->name, target_func);
- WARN_ONCE(worker && ((worker->current_pwq->wq->flags &
+ WARN_ONCE(worker && worker->current_pwq && ((worker->current_pwq->wq->flags &
(WQ_MEM_RECLAIM | __WQ_LEGACY)) == WQ_MEM_RECLAIM),
"workqueue: WQ_MEM_RECLAIM %s:%ps is flushing !WQ_MEM_RECLAIM %s:%ps",
worker->current_pwq->wq->name, worker->current_func,