diff options
Diffstat (limited to 'kernel')
| -rw-r--r-- | kernel/bpf/btf.c | 136 | ||||
| -rw-r--r-- | kernel/bpf/check_btf.c | 12 | ||||
| -rw-r--r-- | kernel/bpf/core.c | 21 | ||||
| -rw-r--r-- | kernel/bpf/crypto.c | 5 | ||||
| -rw-r--r-- | kernel/bpf/fixups.c | 24 | ||||
| -rw-r--r-- | kernel/bpf/hashtab.c | 43 | ||||
| -rw-r--r-- | kernel/bpf/helpers.c | 60 | ||||
| -rw-r--r-- | kernel/bpf/memalloc.c | 50 | ||||
| -rw-r--r-- | kernel/bpf/offload.c | 2 | ||||
| -rw-r--r-- | kernel/bpf/states.c | 16 | ||||
| -rw-r--r-- | kernel/bpf/syscall.c | 15 | ||||
| -rw-r--r-- | kernel/bpf/verifier.c | 162 | ||||
| -rw-r--r-- | kernel/cgroup/cpuset.c | 3 | ||||
| -rw-r--r-- | kernel/cgroup/pids.c | 5 | ||||
| -rw-r--r-- | kernel/events/core.c | 17 | ||||
| -rw-r--r-- | kernel/exit.c | 28 | ||||
| -rw-r--r-- | kernel/fork.c | 2 | ||||
| -rw-r--r-- | kernel/kprobes.c | 22 | ||||
| -rw-r--r-- | kernel/power/hibernate.c | 26 | ||||
| -rw-r--r-- | kernel/sched/core.c | 2 | ||||
| -rw-r--r-- | kernel/sched/ext/cid.c | 26 | ||||
| -rw-r--r-- | kernel/sched/ext/ext.c | 156 | ||||
| -rw-r--r-- | kernel/sched/ext/inlines.h | 4 | ||||
| -rw-r--r-- | kernel/sched/ext/internal.h | 60 | ||||
| -rw-r--r-- | kernel/sched/ext/sub.c | 11 | ||||
| -rw-r--r-- | kernel/sched/ext/types.h | 4 | ||||
| -rw-r--r-- | kernel/sched/fair.c | 417 | ||||
| -rw-r--r-- | kernel/sched/topology.c | 22 | ||||
| -rw-r--r-- | kernel/trace/fprobe.c | 15 | ||||
| -rw-r--r-- | kernel/workqueue.c | 2 |
30 files changed, 963 insertions, 405 deletions
diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c index 9f33e95d5741..d870bc5e50bc 100644 --- a/kernel/bpf/btf.c +++ b/kernel/bpf/btf.c @@ -4268,13 +4268,10 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec) { int i; - /* There are three types that signify ownership of some other type: - * kptr_ref, bpf_list_head, bpf_rb_root. - * kptr_ref only supports storing kernel types, which can't store - * references to program allocated local types. - * - * Hence we only need to ensure that bpf_{list_head,rb_root} ownership - * does not form cycles. + /* + * Check fields which require the complete BTF and initialize runtime + * metadata. Ownership relationships are validated after every record has + * been fixed up. */ if (IS_ERR_OR_NULL(rec) || !(rec->field_mask & (BPF_GRAPH_ROOT | BPF_UPTR))) return 0; @@ -4305,51 +4302,88 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec) if (!meta) return -EFAULT; rec->fields[i].graph_root.value_rec = meta->record; + } + return 0; +} - /* We need to set value_rec for all root types, but no need - * to check ownership cycle for a type unless it's also a - * node type. - */ - if (!(rec->field_mask & BPF_GRAPH_NODE)) +static int btf_owned_type_idx(const struct btf *btf, struct btf_struct_metas *tab, + const struct btf_field *field) +{ + struct btf_struct_meta *meta; + u32 btf_id; + + if (field->type & BPF_GRAPH_ROOT) { + btf_id = field->graph_root.value_btf_id; + } else if (field->type == BPF_KPTR_REF || field->type == BPF_KPTR_PERCPU) { + if (btf_is_kernel(field->kptr.btf)) + return -ENOENT; + btf_id = field->kptr.btf_id; + } else { + return -ENOENT; + } + + meta = btf_find_struct_meta(btf, btf_id); + if (!meta) + return field->type & BPF_GRAPH_ROOT ? -EFAULT : -ENOENT; + return meta - tab->types; +} + +/* + * Each ownership edge adds kernel frames through bpf_obj_free_fields() and + * __bpf_obj_drop_impl(). Keep the bound deliberately small because object + * destruction can itself run below a BPF call chain. A final pointee without + * special fields is not present in the struct metadata table and adds only a + * non-recursing drop. + */ +#define BTF_MAX_OWNERSHIP_DEPTH 8 + +static int btf_ownership_depth(const struct btf *btf, + struct btf_struct_metas *tab, u8 *depth, + int idx, int depth_left) +{ + const struct btf_record *rec = tab->types[idx].record; + int i, ret, max_depth = 0; + + if (!depth_left) + return -ELOOP; + if (depth[idx]) + goto done; + + for (i = 0; i < rec->cnt; i++) { + ret = btf_owned_type_idx(btf, tab, &rec->fields[i]); + if (ret == -ENOENT) continue; + if (ret < 0) + return ret; + ret = btf_ownership_depth(btf, tab, depth, ret, depth_left - 1); + if (ret < 0) + return ret; + max_depth = max(max_depth, ret); + } + depth[idx] = max_depth + 1; +done: + return depth[idx] > depth_left ? -ELOOP : depth[idx]; +} - /* We need to ensure ownership acyclicity among all types. The - * proper way to do it would be to topologically sort all BTF - * IDs based on the ownership edges, since there can be multiple - * bpf_{list_head,rb_node} in a type. Instead, we use the - * following resaoning: - * - * - A type can only be owned by another type in user BTF if it - * has a bpf_{list,rb}_node. Let's call these node types. - * - A type can only _own_ another type in user BTF if it has a - * bpf_{list_head,rb_root}. Let's call these root types. - * - * We ensure that if a type is both a root and node, its - * element types cannot be root types. - * - * To ensure acyclicity: - * - * When A is an root type but not a node, its ownership - * chain can be: - * A -> B -> C - * Where: - * - A is an root, e.g. has bpf_rb_root. - * - B is both a root and node, e.g. has bpf_rb_node and - * bpf_list_head. - * - C is only an root, e.g. has bpf_list_node - * - * When A is both a root and node, some other type already - * owns it in the BTF domain, hence it can not own - * another root type through any of the ownership edges. - * A -> B - * Where: - * - A is both an root and node. - * - B is only an node. - */ - if (meta->record->field_mask & BPF_GRAPH_ROOT) - return -ELOOP; +static int btf_check_ownership_depth(const struct btf *btf, + struct btf_struct_metas *tab) +{ + u8 *depth; + int i, ret = 0; + + depth = kvcalloc(tab->cnt, sizeof(*depth), GFP_KERNEL | __GFP_NOWARN); + if (!depth) + return -ENOMEM; + + for (i = 0; i < tab->cnt; i++) { + ret = btf_ownership_depth(btf, tab, depth, i, + BTF_MAX_OWNERSHIP_DEPTH); + if (ret < 0) + break; + ret = 0; } - return 0; + kvfree(depth); + return ret; } static void __btf_struct_show(const struct btf *btf, const struct btf_type *t, @@ -6044,6 +6078,10 @@ static struct btf *btf_parse(const union bpf_attr *attr, bpfptr_t uattr, if (err < 0) goto errout_meta; } + + err = btf_check_ownership_depth(btf, struct_meta_tab); + if (err < 0) + goto errout_meta; } err = bpf_log_attr_finalize(attr_log, &env->log); @@ -7187,7 +7225,7 @@ again: if (btf_type_is_int(t)) return WALK_SCALAR; - if (!btf_type_is_struct(t)) + if (!btf_type_is_struct(t) || !t->size) goto error; off = (off - moff) % t->size; diff --git a/kernel/bpf/check_btf.c b/kernel/bpf/check_btf.c index 0e8b3ccc7a5b..4c1ed842f661 100644 --- a/kernel/bpf/check_btf.c +++ b/kernel/bpf/check_btf.c @@ -338,9 +338,9 @@ err_free: #define MIN_CORE_RELO_SIZE sizeof(struct bpf_core_relo) #define MAX_CORE_RELO_SIZE MAX_FUNCINFO_REC_SIZE -static int check_core_relo(struct bpf_verifier_env *env, - const union bpf_attr *attr, - bpfptr_t uattr) +int bpf_check_core_relo(struct bpf_verifier_env *env, + const union bpf_attr *attr, + bpfptr_t uattr) { u32 i, nr_core_relo, ncopy, expected_size, rec_size; struct bpf_core_relo core_relo = {}; @@ -414,7 +414,7 @@ int bpf_prepare_btf_info(struct bpf_verifier_env *env, struct btf *btf; int err; - if (!attr->func_info_cnt && !attr->line_info_cnt) { + if (!attr->func_info_cnt && !attr->line_info_cnt && !attr->core_relo_cnt) { if (check_abnormal_return(env)) return -EINVAL; return 0; @@ -455,9 +455,5 @@ int bpf_check_btf_info(struct bpf_verifier_env *env, if (err) return err; - err = check_core_relo(env, attr, uattr); - if (err) - return err; - return 0; } diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c index 8b294dfc1ad4..2e3bf8113ae9 100644 --- a/kernel/bpf/core.c +++ b/kernel/bpf/core.c @@ -19,6 +19,7 @@ #include <uapi/linux/btf.h> #include <linux/filter.h> +#include <linux/sched/signal.h> #include <linux/skbuff.h> #include <linux/static_call.h> #include <linux/vmalloc.h> @@ -1619,6 +1620,8 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp * fix it up here on error. */ bpf_jit_prog_release_other(prog, clone); + if (env && fatal_signal_pending(current)) + return ERR_PTR(-EINTR); return IS_ERR(tmp) ? tmp : ERR_PTR(-ENOMEM); } @@ -2636,11 +2639,14 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc orig_prog = prog; prog = bpf_jit_blind_constants(env, prog); /* - * If blinding was requested and we failed during blinding, we must fall - * back to the interpreter. + * Fall back to the interpreter after blinding failures, except when + * the loader was killed. */ - if (IS_ERR(prog)) + if (IS_ERR(prog)) { + if (PTR_ERR(prog) == -EINTR) + return prog; goto out_restore; + } prog = bpf_int_jit_compile(env, prog); if (prog->jited) { @@ -2659,6 +2665,8 @@ out_restore: struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct bpf_prog *fp, int *err) { + struct bpf_prog *jit_prog; + /* In case of BPF to BPF calls, verifier did all the prep * work with regards to JITing, etc. */ @@ -2681,7 +2689,12 @@ struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct if (*err) return fp; - fp = bpf_prog_jit_compile(env, fp); + jit_prog = bpf_prog_jit_compile(env, fp); + if (IS_ERR(jit_prog)) { + *err = PTR_ERR(jit_prog); + return fp; + } + fp = jit_prog; bpf_prog_jit_attempt_done(fp); if (!fp->jited && jit_needed) { *err = -ENOTSUPP; diff --git a/kernel/bpf/crypto.c b/kernel/bpf/crypto.c index 51f89cecefb4..3f3fe2450fc6 100644 --- a/kernel/bpf/crypto.c +++ b/kernel/bpf/crypto.c @@ -149,8 +149,9 @@ bpf_crypto_ctx_create(const struct bpf_crypto_params *params, u32 params__sz, const struct bpf_crypto_type *type; struct bpf_crypto_ctx *ctx; - if (!params || params->reserved[0] || params->reserved[1] || - params__sz != sizeof(struct bpf_crypto_params)) { + if (!params || + params__sz != sizeof(struct bpf_crypto_params) || + params->reserved[0] || params->reserved[1]) { *err = -EINVAL; return NULL; } diff --git a/kernel/bpf/fixups.c b/kernel/bpf/fixups.c index 52d3cec33672..d6f83521fc78 100644 --- a/kernel/bpf/fixups.c +++ b/kernel/bpf/fixups.c @@ -8,6 +8,7 @@ #include <linux/bsearch.h> #include <linux/sort.h> #include <linux/perf_event.h> +#include <linux/sched/signal.h> #include <net/xdp.h> #include "disasm.h" @@ -306,12 +307,28 @@ static void adjust_poke_descs(struct bpf_prog *prog, u32 off, u32 len) } } +/* + * Some post-verification instruction rewriting passes require an + * O(prog->len) operation per instruction. Keep their shared primitives + * killable and preemptible. + */ +static bool bpf_rewrite_must_abort(void) +{ + if (fatal_signal_pending(current)) + return true; + cond_resched(); + return false; +} + struct bpf_prog *bpf_patch_insn_data(struct bpf_verifier_env *env, u32 off, const struct bpf_insn *patch, u32 len) { struct bpf_prog *new_prog; struct bpf_insn_aux_data *new_data = NULL; + if (bpf_rewrite_must_abort()) + return NULL; + if (len > 1) { new_data = vrealloc(env->insn_aux_data, array_size(env->prog->len + len - 1, @@ -523,6 +540,9 @@ static int verifier_remove_insns(struct bpf_verifier_env *env, u32 off, u32 cnt) unsigned int orig_prog_len = env->prog->len; int err; + if (bpf_rewrite_must_abort()) + return -EINTR; + if (bpf_prog_is_offloaded(env->prog->aux)) bpf_prog_offload_remove_insns(env, off, cnt); @@ -1356,7 +1376,7 @@ int bpf_jit_subprogs(struct bpf_verifier_env *env) } prog = bpf_jit_blind_constants(env, prog); if (IS_ERR(prog)) { - err = -ENOMEM; + err = PTR_ERR(prog); prog = orig_prog; goto out_restore; } @@ -1433,7 +1453,7 @@ int bpf_fixup_call_args(struct bpf_verifier_env *env) err = bpf_jit_subprogs(env); if (err == 0) return 0; - if (err == -EFAULT) + if (err == -EFAULT || err == -EINTR) return err; } #ifndef CONFIG_BPF_JIT_ALWAYS_ON diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c index 6f331c80130d..f9464e566f10 100644 --- a/kernel/bpf/hashtab.c +++ b/kernel/bpf/hashtab.c @@ -1054,14 +1054,17 @@ static void pcpu_init_value(struct bpf_htab *htab, void __percpu *pptr, /* When not setting the initial value on all cpus, zero-fill element * values for other cpus. Otherwise, bpf program has no way to ensure * known initial values for cpus other than current one - * (onallcpus=false always when coming from bpf prog). + * (onallcpus=false always when coming from bpf prog, + * map_flags & BPF_F_CPU when coming from syscall but setting + * only one cpu). */ - if (!onallcpus) { - int current_cpu = raw_smp_processor_id(); + if (!onallcpus || (map_flags & BPF_F_CPU)) { + int init_cpu = (map_flags & BPF_F_CPU) ? map_flags >> 32 : + raw_smp_processor_id(); int cpu; for_each_possible_cpu(cpu) { - if (cpu == current_cpu) + if (cpu == init_cpu) copy_map_value(&htab->map, per_cpu_ptr(pptr, cpu), value); else /* Since elem is preallocated, we cannot touch special fields */ zero_map_value(&htab->map, per_cpu_ptr(pptr, cpu)); @@ -1772,6 +1775,12 @@ static int htab_lru_percpu_map_lookup_and_delete_elem(struct bpf_map *map, flags); } +/* + * Max consecutive empty buckets to walk in one RCU + + * instrumentation-disabled section before rescheduling. + */ +#define HTAB_BATCH_EMPTY_RESCHED 64 + static int __htab_map_lookup_and_delete_batch(struct bpf_map *map, const union bpf_attr *attr, @@ -1793,6 +1802,7 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map, unsigned long flags = 0; bool locked = false; struct htab_elem *l; + u32 empty_cnt = 0; struct bucket *b; int ret = 0; @@ -1971,30 +1981,41 @@ again_nocopy: } next_batch: - /* If we are not copying data, we can go to next bucket and avoid - * unlocking the rcu. + /* + * If we are not copying data, we can go to next bucket and avoid + * unlocking the rcu. Bound the walk though: after + * HTAB_BATCH_EMPTY_RESCHED consecutive empty buckets, fully exit + * the critical section (no locks are held here) and reschedule. */ if (!bucket_cnt && (batch + 1 < htab->n_buckets)) { batch++; - goto again_nocopy; + if (++empty_cnt < HTAB_BATCH_EMPTY_RESCHED) + goto again_nocopy; + empty_cnt = 0; + rcu_read_unlock(); + bpf_enable_instrumentation(); + cond_resched_tasks_rcu_qs(); + goto again; } rcu_read_unlock(); bpf_enable_instrumentation(); - if (bucket_cnt && (copy_to_user(ukeys + total * key_size, keys, - key_size * bucket_cnt) || - copy_to_user(uvalues + total * value_size, values, - value_size * bucket_cnt))) { + if (bucket_cnt && (copy_to_user(ukeys + (size_t)total * key_size, keys, + (size_t)key_size * bucket_cnt) || + copy_to_user(uvalues + (size_t)total * value_size, values, + (size_t)value_size * bucket_cnt))) { ret = -EFAULT; goto after_loop; } total += bucket_cnt; + empty_cnt = 0; batch++; if (batch >= htab->n_buckets) { ret = -ENOENT; goto after_loop; } + cond_resched_tasks_rcu_qs(); goto again; after_loop: diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c index b3cc5c8fc875..712dca5a2c5b 100644 --- a/kernel/bpf/helpers.c +++ b/kernel/bpf/helpers.c @@ -4883,7 +4883,7 @@ BTF_ID(func, bpf_cgroup_release_dtor) BTF_KFUNCS_START(common_btf_ids) BTF_ID_FLAGS(func, bpf_cast_to_kern_ctx, KF_FASTCALL) -BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL) +BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL | KF_PERFMON) BTF_ID_FLAGS(func, bpf_rcu_read_lock) BTF_ID_FLAGS(func, bpf_rcu_read_unlock) BTF_ID_FLAGS(func, bpf_dynptr_slice, KF_RET_NULL) @@ -4920,26 +4920,26 @@ BTF_ID_FLAGS(func, bpf_wq_set_callback, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_wq_start) BTF_ID_FLAGS(func, bpf_preempt_disable) BTF_ID_FLAGS(func, bpf_preempt_enable) -BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW) +BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW | KF_PERFMON) BTF_ID_FLAGS(func, bpf_iter_bits_next, KF_ITER_NEXT | KF_RET_NULL) BTF_ID_FLAGS(func, bpf_iter_bits_destroy, KF_ITER_DESTROY) -BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_get_kmem_cache) +BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_get_kmem_cache, KF_PERFMON) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_new, KF_ITER_NEW | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_destroy, KF_ITER_DESTROY | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_local_irq_save) BTF_ID_FLAGS(func, bpf_local_irq_restore) #ifdef CONFIG_BPF_EVENTS -BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr) -BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE) +BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE | KF_PERFMON) #endif #ifdef CONFIG_DMA_SHARED_BUFFER BTF_ID_FLAGS(func, bpf_iter_dmabuf_new, KF_ITER_NEW | KF_SLEEPABLE) @@ -4947,26 +4947,26 @@ BTF_ID_FLAGS(func, bpf_iter_dmabuf_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPAB BTF_ID_FLAGS(func, bpf_iter_dmabuf_destroy, KF_ITER_DESTROY | KF_SLEEPABLE) #endif BTF_ID_FLAGS(func, __bpf_trap) -BTF_ID_FLAGS(func, bpf_strcmp); -BTF_ID_FLAGS(func, bpf_strcasecmp); -BTF_ID_FLAGS(func, bpf_strncasecmp); -BTF_ID_FLAGS(func, bpf_strchr); -BTF_ID_FLAGS(func, bpf_strchrnul); -BTF_ID_FLAGS(func, bpf_strnchr); -BTF_ID_FLAGS(func, bpf_strrchr); -BTF_ID_FLAGS(func, bpf_strlen); -BTF_ID_FLAGS(func, bpf_strnlen); -BTF_ID_FLAGS(func, bpf_strspn); -BTF_ID_FLAGS(func, bpf_strcspn); -BTF_ID_FLAGS(func, bpf_strstr); -BTF_ID_FLAGS(func, bpf_strcasestr); -BTF_ID_FLAGS(func, bpf_strnstr); -BTF_ID_FLAGS(func, bpf_strncasestr); +BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcasecmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strncasecmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strchrnul, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strrchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strlen, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnlen, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strspn, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcspn, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strstr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcasestr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnstr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strncasestr, KF_PERFMON); #if defined(CONFIG_BPF_LSM) && defined(CONFIG_CGROUPS) BTF_ID_FLAGS(func, bpf_cgroup_read_xattr, KF_RCU) #endif -BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE) -BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE) +BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON) BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_dynptr_from_file) diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c index e9662db7198f..8a8f088e83e6 100644 --- a/kernel/bpf/memalloc.c +++ b/kernel/bpf/memalloc.c @@ -119,6 +119,7 @@ struct bpf_mem_cache { struct llist_head waiting_for_gp_ttrace; struct rcu_head rcu_ttrace; atomic_t call_rcu_ttrace_in_progress; + raw_spinlock_t lock; }; struct bpf_mem_caches { @@ -214,25 +215,24 @@ static void alloc_bulk(struct bpf_mem_cache *c, int cnt, int node, bool atomic) gfp = __GFP_NOWARN | __GFP_ACCOUNT; gfp |= atomic ? GFP_NOWAIT : GFP_KERNEL; - for (i = 0; i < cnt; i++) { - /* - * For every 'c' llist_del_first(&c->free_by_rcu_ttrace); is - * done only by one CPU == current CPU. Other CPUs might - * llist_add() and llist_del_all() in parallel. - */ - obj = llist_del_first(&c->free_by_rcu_ttrace); - if (!obj) - break; - add_obj_to_free_list(c, obj); - } - if (i >= cnt) - return; + /* + * c->lock serializes concurrent llist_del_first() against + * llist_del_all() in __free_rcu() and do_call_rcu_ttrace(). + */ + scoped_guard(raw_spinlock_irqsave, &c->lock) { + for (i = 0; i < cnt; i++) { + obj = llist_del_first(&c->free_by_rcu_ttrace); + if (!obj) + break; + add_obj_to_free_list(c, obj); + } - for (; i < cnt; i++) { - obj = llist_del_first(&c->waiting_for_gp_ttrace); - if (!obj) - break; - add_obj_to_free_list(c, obj); + for (; i < cnt; i++) { + obj = llist_del_first(&c->waiting_for_gp_ttrace); + if (!obj) + break; + add_obj_to_free_list(c, obj); + } } if (i >= cnt) return; @@ -279,8 +279,12 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per static void __free_rcu(struct rcu_head *head) { struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace); + struct llist_node *llnode; + + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->waiting_for_gp_ttrace); - free_all(c, llist_del_all(&c->waiting_for_gp_ttrace), !!c->percpu_size); + free_all(c, llnode, !!c->percpu_size); atomic_set(&c->call_rcu_ttrace_in_progress, 0); } @@ -300,7 +304,8 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c) if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) { if (unlikely(READ_ONCE(c->draining))) { - llnode = llist_del_all(&c->free_by_rcu_ttrace); + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->free_by_rcu_ttrace); free_all(c, llnode, !!c->percpu_size); } return; @@ -535,6 +540,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } @@ -557,7 +563,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; - + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } @@ -609,7 +615,7 @@ int bpf_mem_alloc_percpu_unit_init(struct bpf_mem_alloc *ma, int size) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; - + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } diff --git a/kernel/bpf/offload.c b/kernel/bpf/offload.c index 0d6f5569588c..d855399812ee 100644 --- a/kernel/bpf/offload.c +++ b/kernel/bpf/offload.c @@ -698,6 +698,8 @@ static bool __bpf_offload_dev_match(struct bpf_prog *prog, return false; if (offload->netdev == netdev) return true; + if (!bpf_prog_is_offloaded(prog->aux)) + return false; ondev1 = bpf_offload_find_netdev(offload->netdev); ondev2 = bpf_offload_find_netdev(netdev); diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c index 66fb11b6c6a7..012b82513a3b 100644 --- a/kernel/bpf/states.c +++ b/kernel/bpf/states.c @@ -491,7 +491,8 @@ static bool regs_exact(const struct bpf_reg_state *rold, { return memcmp(rold, rcur, offsetof(struct bpf_reg_state, id)) == 0 && check_ids(rold->id, rcur->id, idmap) && - check_ids(rold->parent_id, rcur->parent_id, idmap); + check_ids(rold->parent_id, rcur->parent_id, idmap) && + check_ids(rold->map_uid, rcur->map_uid, idmap); } enum exact_level { @@ -616,7 +617,8 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold, range_within(rold, rcur) && tnum_in(rold->var_off, rcur->var_off) && check_ids(rold->id, rcur->id, idmap) && - check_ids(rold->parent_id, rcur->parent_id, idmap); + check_ids(rold->parent_id, rcur->parent_id, idmap) && + check_ids(rold->map_uid, rcur->map_uid, idmap); case PTR_TO_PACKET_META: case PTR_TO_PACKET: /* We must have at least as much range as the old ptr @@ -635,14 +637,14 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold, /* id relations must be preserved */ if (!check_ids(rold->id, rcur->id, idmap)) return false; + /* Preserve displacements between pointers sharing an ID. */ + if (rold->id && rold->r64.base != rcur->r64.base) + return false; /* new val must satisfy old val knowledge */ return range_within(rold, rcur) && tnum_in(rold->var_off, rcur->var_off); case PTR_TO_STACK: - /* two stack pointers are equal only if they're pointing to - * the same stack frame, since fp-8 in foo != fp-8 in bar - */ - return regs_exact(rold, rcur, idmap) && rold->frameno == rcur->frameno; + return regs_exact(rold, rcur, idmap); case PTR_TO_ARENA: return true; case PTR_TO_INSN: @@ -1121,7 +1123,7 @@ static bool states_maybe_looping(struct bpf_verifier_state *old, fcur = cur->frame[fr]; for (i = 0; i < MAX_BPF_REG; i++) if (memcmp(&fold->regs[i], &fcur->regs[i], - offsetof(struct bpf_reg_state, frameno))) + offsetof(struct bpf_reg_state, precise))) return false; return true; } diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c index c7bc9ba9b331..244a939b9d2d 100644 --- a/kernel/bpf/syscall.c +++ b/kernel/bpf/syscall.c @@ -2036,7 +2036,7 @@ int generic_map_delete_batch(struct bpf_map *map, for (cp = 0; cp < max_count; cp++) { err = -EFAULT; - if (copy_from_user(key, keys + cp * map->key_size, + if (copy_from_user(key, keys + (size_t)cp * map->key_size, map->key_size)) break; @@ -2098,9 +2098,9 @@ int generic_map_update_batch(struct bpf_map *map, struct file *map_file, for (cp = 0; cp < max_count; cp++) { err = -EFAULT; - if (copy_from_user(key, keys + cp * map->key_size, + if (copy_from_user(key, keys + (size_t)cp * map->key_size, map->key_size) || - copy_from_user(value, values + cp * value_size, value_size)) + copy_from_user(value, values + (size_t)cp * value_size, value_size)) break; err = bpf_map_update_value(map, map_file, key, value, @@ -2179,12 +2179,12 @@ int generic_map_lookup_batch(struct bpf_map *map, if (err) goto free_buf; - if (copy_to_user(keys + cp * map->key_size, key, + if (copy_to_user(keys + (size_t)cp * map->key_size, key, map->key_size)) { err = -EFAULT; goto free_buf; } - if (copy_to_user(values + cp * value_size, value, value_size)) { + if (copy_to_user(values + (size_t)cp * value_size, value, value_size)) { err = -EFAULT; goto free_buf; } @@ -6042,7 +6042,10 @@ struct bpf_link *bpf_link_get_curr_or_next(u32 *id) again: link = idr_get_next(&link_idr, id); if (link) { - link = bpf_link_inc_not_zero(link); + if (link->id) + link = bpf_link_inc_not_zero(link); + else + link = ERR_PTR(-EAGAIN); if (IS_ERR(link)) { (*id)++; goto again; diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c index 72a3f5998dd2..41b49c56e123 100644 --- a/kernel/bpf/verifier.c +++ b/kernel/bpf/verifier.c @@ -567,7 +567,7 @@ static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_s } off = reg->var_off.value; - if (off % BPF_REG_SIZE) { + if (off >= 0 || off % BPF_REG_SIZE) { verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off); return -EINVAL; } @@ -1864,6 +1864,7 @@ static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm) offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type)); reg->id = 0; reg->parent_id = 0; + reg->map_uid = 0; ___mark_reg_known(reg, imm); } @@ -1925,17 +1926,18 @@ static void refine_map_lookup_value(struct bpf_reg_state *reg) if (map->inner_map_meta) { reg->type = CONST_PTR_TO_MAP | maybe_null; reg->map_ptr = map->inner_map_meta; - /* transfer reg's id which is unique for every map_lookup_elem + /* + * transfer reg's id which is unique for every map_lookup_elem * as UID of the inner map. */ - if (btf_record_has_field(map->inner_map_meta->record, - BPF_TIMER | BPF_WORKQUEUE | BPF_TASK_WORK)) - reg->map_uid = reg->id; + reg->map_uid = reg->id; } else if (map->map_type == BPF_MAP_TYPE_XSKMAP) { reg->type = PTR_TO_XDP_SOCK | maybe_null; + reg->map_uid = 0; } else if (map->map_type == BPF_MAP_TYPE_SOCKMAP || map->map_type == BPF_MAP_TYPE_SOCKHASH) { reg->type = PTR_TO_SOCKET | maybe_null; + reg->map_uid = 0; } } @@ -3042,6 +3044,8 @@ static int check_subprogs(struct bpf_verifier_env *env) subprog[cur_subprog].exit_idx = i; goto next; } + if (insn_is_gotox(&insn[i])) + goto next; off = i + bpf_jmp_offset(&insn[i]) + 1; if (off < subprog_start || off >= subprog_end) { verbose(env, "jump out of range from insn %d to %d\n", i, off); @@ -3061,7 +3065,8 @@ next: */ if (code != (BPF_JMP | BPF_EXIT) && code != (BPF_JMP32 | BPF_JA) && - code != (BPF_JMP | BPF_JA)) { + code != (BPF_JMP | BPF_JA) && + !insn_is_gotox(&insn[i])) { verbose(env, "last insn is not an exit or jmp\n"); bpf_diag_program_structure( env, i, "subprogram can fall through", @@ -3582,7 +3587,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env, save_register_state(env, state, spi, reg, size); /* Break the relation on a narrowing spill. */ if (!reg_value_fits) - state->stack[spi].spilled_ptr.id = 0; + clear_scalar_id(&state->stack[spi].spilled_ptr); } else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) && env->bpf_capable) { struct bpf_reg_state *tmp_reg = &env->fake_reg[0]; @@ -6453,6 +6458,15 @@ static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct b return -EACCES; } + if (rdonly_untrusted && !env->allow_ptr_leaks) { + verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", + reg_type_str(env, reg->type)); + bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory", + "the access requires CAP_PERFMON", + "Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers."); + return -EPERM; + } + /* * Accesses to untrusted PTR_TO_MEM are done through probe * instructions, hence no need to check bounds in that case. @@ -9736,6 +9750,16 @@ static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog, if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL, NULL)) return -EINVAL; + /* + * PTR_TO_PACKET get passed as PTR_TO_MEM, preventing + * us from adjusting bounds tracking info. + */ + if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) && + sub->changes_pkt_data) { + bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n", + reg_arg_name(env, argno), subprog); + return -EINVAL; + } if (!(arg->arg_type & PTR_MAYBE_NULL) && (type_may_be_null(reg->type) || bpf_register_is_null(reg))) { bpf_log(log, "%s is expected to be non-NULL\n", @@ -9919,6 +9943,7 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn, if (err == -EFAULT) return err; if (bpf_subprog_is_global(env, subprog)) { + struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog); const char *sub_name = bpf_subprog_name(env, subprog); const char *operation; bool returns_void; @@ -9950,11 +9975,10 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn, if (env->log.level & BPF_LOG_LEVEL) verbose(env, "Func#%d ('%s') is global and assumed valid.\n", subprog, sub_name); + sub_aux->called[in_sleepable_context(env)] = true; returns_void = subprog_returns_void(env, subprog); if (env->subprog_info[subprog].changes_pkt_data) clear_all_pkt_pointers(env); - /* mark global subprog for verifying after main prog */ - subprog_aux(env, subprog)->called = true; if (returns_void) bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED); else @@ -10036,6 +10060,7 @@ int map_set_for_each_callback_args(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr; callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* pointer to stack or null */ callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3]; @@ -10132,6 +10157,7 @@ static int set_timer_callback_state(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = map_ptr; callee->regs[BPF_REG_3].map_uid = map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* unused */ bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); @@ -10250,6 +10276,7 @@ static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = map_ptr; callee->regs[BPF_REG_3].map_uid = map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* unused */ bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); @@ -10772,11 +10799,7 @@ int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id, /* Check if we're in a sleepable context. */ static inline bool in_sleepable_context(struct bpf_verifier_env *env) { - return !env->cur_state->active_rcu_locks && - !env->cur_state->active_preempt_locks && - !env->cur_state->active_locks && - !env->cur_state->active_irq_id && - in_sleepable(env); + return !in_rcu_cs(env); } static const char *non_sleepable_context_description(struct bpf_verifier_env *env) @@ -11356,6 +11379,11 @@ static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta) return meta->kfunc_flags & KF_DESTRUCTIVE; } +static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta) +{ + return meta->kfunc_flags & KF_PERFMON; +} + static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta) { return meta->kfunc_flags & KF_RCU; @@ -13834,6 +13862,15 @@ static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn, return -EACCES; } + if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) { + verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", + func_name); + operation = bpf_diag_fmt(env, "kfunc %s", func_name); + bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON", + "Load the program with CAP_PERFMON, or avoid the kfunc."); + return -EPERM; + } + sleepable = bpf_is_kfunc_sleepable(&meta); if (sleepable && !in_sleepable(env)) { verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name); @@ -15704,6 +15741,7 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg; struct bpf_reg_state *ptr_reg = NULL, off_reg = {0}; bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64); + struct bpf_insn_aux_data *aux = cur_aux(env); u8 opcode = BPF_OP(insn->code); int err; @@ -15715,13 +15753,24 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, /* Case where at least one operand is an arena. */ if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) { - struct bpf_insn_aux_data *aux = cur_aux(env); if (dst_reg->type != PTR_TO_ARENA) *dst_reg = *src_reg; if (BPF_CLASS(insn->code) == BPF_ALU64) { /* + * Only arena pointers set needs_zext, but doing so + * modifies the instruction at fixup time to an ALU32 + * and makes it unsuitable for 64-bit scalar args. We + * prevent zext from being set if the instruction has + * been previously called with non-arena registers. + */ + if (aux->prevent_zext) { + verbose(env, "same insn cannot be used with and without arena pointer\n"); + return -EINVAL; + } + + /* * 32-bit operations zero upper bits automatically. * 64-bit operations need to be converted to 32. */ @@ -15733,6 +15782,16 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, return 0; } + /* Prevent the instruction from being used with arena pointers (see above). */ + if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) { + if (aux->needs_zext) { + verbose(env, "same insn cannot be used with and without arena pointer\n"); + return -EINVAL; + } + + aux->prevent_zext = true; + } + if (dst_reg->type != SCALAR_VALUE) ptr_reg = dst_reg; @@ -19421,13 +19480,14 @@ static void free_states(struct bpf_verifier_env *env) } } -static int do_check_common(struct bpf_verifier_env *env, int subprog) +static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable) { bool pop_log = !(env->log.level & BPF_LOG_LEVEL2); struct bpf_subprog_info *sub = subprog_info(env, subprog); struct bpf_prog_aux *aux = env->prog->aux; struct bpf_verifier_state *state; struct bpf_reg_state *regs; + u32 old_insns_total = sub->insns_total; u32 insn_processed = env->insn_processed; int ret, i; @@ -19440,7 +19500,7 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog) state->curframe = 0; state->speculative = false; state->branches = 1; - state->in_sleepable = env->prog->sleepable; + state->in_sleepable = is_sleepable; state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT); if (!state->frame[0]) { kfree(state); @@ -19581,8 +19641,10 @@ out: * not accounted as callees by account_current_path(). * Accumulate their total counts as total counts of the main or * global subprog hosting the async call. + * Start from the saved total of earlier contexts: adding to the current + * total would count this pass's synchronous paths twice. */ - env->subprog_info[subprog].insns_total = env->insn_processed - insn_processed; + sub->insns_total = old_insns_total + (env->insn_processed - insn_processed); return ret; } @@ -19610,14 +19672,19 @@ static int do_check_subprogs(struct bpf_verifier_env *env) { struct bpf_prog_aux *aux = env->prog->aux; struct bpf_func_info_aux *sub_aux; - int i, ret, new_cnt; + int context, i, ret, new_cnt; if (!aux->func_info) return 0; - /* exception callback is presumed to be always called */ - if (env->exception_callback_subprog) - subprog_aux(env, env->exception_callback_subprog)->called = true; + /* + * Callbacks cannot throw, so the exception callback always runs in the + * main program's context. It is presumed to be always called. + */ + if (env->exception_callback_subprog) { + sub_aux = subprog_aux(env, env->exception_callback_subprog); + sub_aux->called[env->prog->sleepable] = true; + } again: new_cnt = 0; @@ -19626,29 +19693,28 @@ again: continue; sub_aux = subprog_aux(env, i); - if (!sub_aux->called || sub_aux->verified) - continue; + for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) { + if (!sub_aux->called[context] || sub_aux->verified[context]) + continue; - env->insn_idx = env->subprog_info[i].start; - WARN_ON_ONCE(env->insn_idx == 0); - ret = do_check_common(env, i); - if (ret) { - return ret; - } else if (env->log.level & BPF_LOG_LEVEL) { - verbose(env, "Func#%d ('%s') is safe for any args that match its prototype\n", - i, bpf_subprog_name(env, i)); - } + env->insn_idx = env->subprog_info[i].start; + WARN_ON_ONCE(env->insn_idx == 0); + ret = do_check_common(env, i, context); + if (ret) + return ret; + if (env->log.level & BPF_LOG_LEVEL) + verbose(env, "Func#%d ('%s') is safe for any args " + "that match its prototype\n", + i, bpf_subprog_name(env, i)); - /* We verified new global subprog, it might have called some - * more global subprogs that we haven't verified yet, so we - * need to do another pass over subprogs to verify those. - */ - sub_aux->verified = true; - new_cnt++; + sub_aux->verified[context] = true; + new_cnt++; + } } - /* We can't loop forever as we verify at least one global subprog on - * each pass. + /* + * We can't loop forever as each pass verifies at least one new context, + * and there are only two contexts per global subprog. */ if (new_cnt) goto again; @@ -19661,7 +19727,7 @@ static int do_check_main(struct bpf_verifier_env *env) int ret; env->insn_idx = 0; - ret = do_check_common(env, 0); + ret = do_check_common(env, 0, env->prog->sleepable); if (!ret) env->prog->aux->stack_depth = env->subprog_info[0].stack_depth; return ret; @@ -21170,6 +21236,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, ret = bpf_diag_init(env); if (ret) goto err_prep; + if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) { + verbose(env, "invalid bpf_ld_imm64 insn\n"); + ret = -EINVAL; + goto err_prep; + } if (env->signature) { ret = bpf_prog_calc_tag(env->prog); if (ret < 0) @@ -21245,6 +21316,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, if (ret < 0) goto skip_full_check; + /* Apply CO-RE before validating the program's instruction layout. */ + ret = bpf_check_core_relo(env, attr, uattr); + if (ret < 0) + goto skip_full_check; + /* Discover all subprograms before validating their layout and BTF. */ ret = add_subprogs(env); if (ret < 0) @@ -21254,7 +21330,7 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, if (ret < 0) goto skip_full_check; - /* Validate BTF against the complete subprogram layout and apply CO-RE. */ + /* Validate BTF against the complete subprogram layout. */ ret = bpf_check_btf_info(env, attr, uattr); if (ret < 0) goto skip_full_check; diff --git a/kernel/cgroup/cpuset.c b/kernel/cgroup/cpuset.c index 2538faac9aba..3f52717c1965 100644 --- a/kernel/cgroup/cpuset.c +++ b/kernel/cgroup/cpuset.c @@ -1591,10 +1591,11 @@ static int remote_partition_enable(struct cpuset *cs, int new_prs, * above it or remote partition root underneath it is not allowed. */ compute_excpus(cs, tmp->new_cpus); - WARN_ON_ONCE(cpumask_intersects(tmp->new_cpus, subpartitions_cpus)); if (!cpumask_intersects(tmp->new_cpus, cpu_active_mask) || cpumask_subset(top_cpuset.effective_cpus, tmp->new_cpus)) return PERR_INVCPUS; + if (cpumask_intersects(tmp->new_cpus, subpartitions_cpus)) + return PERR_NOCPUS; if (((new_prs == PRS_ISOLATED) && !isolated_cpus_can_update(tmp->new_cpus, NULL)) || prstate_housekeeping_conflict(new_prs, tmp->new_cpus)) diff --git a/kernel/cgroup/pids.c b/kernel/cgroup/pids.c index ecbb839d2acb..78cdc0558d0c 100644 --- a/kernel/cgroup/pids.c +++ b/kernel/cgroup/pids.c @@ -253,6 +253,11 @@ static void pids_event(struct pids_cgroup *pids_forking, } if (!cgroup_subsys_on_dfl(pids_cgrp_subsys) || cgrp_dfl_root.flags & CGRP_ROOT_PIDS_LOCAL_EVENTS) { + /* + * pids.events reports the local counter on legacy hierarchies + * and when pids_localevents is enabled. + */ + cgroup_file_notify(&p->events_file); cgroup_file_notify(&p->events_local_file); return; } diff --git a/kernel/events/core.c b/kernel/events/core.c index db7b76d6b68a..634d2ccbab82 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx, list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) { cpc = this_cpc(pmu_ctx->pmu); + if (cpc->task_epc != pmu_ctx) + continue; + if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task) pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in); } @@ -3914,7 +3917,7 @@ static void __perf_pmu_sched_task(struct perf_cpu_pmu_context *cpc, perf_ctx_lock(cpuctx, cpuctx->task_ctx); perf_pmu_disable(pmu); - pmu->sched_task(cpc->task_epc, task, sched_in); + pmu->sched_task(&cpc->epc, task, sched_in); perf_pmu_enable(pmu); perf_ctx_unlock(cpuctx, cpuctx->task_ctx); @@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev, struct task_struct *next, bool sched_in) { - struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context); struct perf_cpu_pmu_context *cpc, *cpc2; - /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */ - if (prev == next || cpuctx->task_ctx) + if (prev == next) return; - list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) + list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) { + if (cpc->task_epc) + continue; + __perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in); + } } static void perf_event_switch(struct task_struct *task, @@ -5454,6 +5459,8 @@ attach_task_ctx_data(struct task_struct *task, struct kmem_cache *ctx_cache, } if (refcount_inc_not_zero(&old->refcount)) { + if (global) + old->global = true; free_perf_ctx_data(cd); /* unused */ return 0; } diff --git a/kernel/exit.c b/kernel/exit.c index 424c44a42a4d..282328d2b4cf 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -551,32 +551,6 @@ void mm_update_next_owner(struct mm_struct *mm) } #endif /* CONFIG_MEMCG */ -#if defined(CONFIG_SCHED_CACHE) && defined(CONFIG_NUMA_BALANCING) -/* - * Subtract the memory footprint of the current task from - * mm. - */ -static void exit_mm_sched_cache(struct mm_struct *mm) -{ - unsigned long fp, sub; - - if (!current->total_numa_faults) - return; - /* - * No lock protection due to performance considerations. - * Make sure mm->sc_stat.footprint does not become - * negative. - */ - fp = READ_ONCE(mm->sc_stat.footprint); - sub = min(fp, current->total_numa_faults); - WRITE_ONCE(mm->sc_stat.footprint, fp - sub); -} -#else -static inline void exit_mm_sched_cache(struct mm_struct *mm) -{ -} -#endif /* CONFIG_SCHED_CACHE CONFIG_NUMA_BALANCING */ - /* * Turn us into a lazy TLB process if we * aren't already.. @@ -589,7 +563,7 @@ static void exit_mm(void) if (!mm) return; - exit_mm_sched_cache(mm); + sched_cache_exit_mm(current); mmap_read_lock(mm); mmgrab_lazy_tlb(mm); diff --git a/kernel/fork.c b/kernel/fork.c index 5ef413368912..10f2d05d816a 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1599,6 +1599,7 @@ static int copy_mm(u64 clone_flags, struct task_struct *tsk) tsk->mm = mm; tsk->active_mm = mm; + sched_cache_fork(tsk); return 0; } @@ -2602,6 +2603,7 @@ bad_fork_cleanup_io: bad_fork_cleanup_namespaces: exit_nsproxy_namespaces(p); bad_fork_cleanup_mm: + sched_cache_fork_cleanup(p); if (p->mm) { mm_clear_owner(p->mm, p); mmput(p->mm); diff --git a/kernel/kprobes.c b/kernel/kprobes.c index 6337da5cab9e..4edd8ca5c657 100644 --- a/kernel/kprobes.c +++ b/kernel/kprobes.c @@ -42,6 +42,7 @@ #include <linux/execmem.h> #include <linux/cleanup.h> #include <linux/wait.h> +#include <linux/wait_bit.h> #include <asm/sections.h> #include <asm/cacheflush.h> @@ -526,7 +527,8 @@ enum { OPTIMIZER_ST_FLUSHING = 2, }; -static DECLARE_COMPLETION(optimizer_completion); +/* Bumped at the end of each kprobe_optimizer() pass, under 'kprobe_mutex' */ +static unsigned long optimizer_passes; #define OPTIMIZE_DELAY 5 @@ -654,9 +656,9 @@ static void kprobe_optimizer(void) do_free_cleaned_kprobes(); } - /* Step 5: Kick optimizer again if needed. But if there is a flush requested, */ - if (completion_done(&optimizer_completion)) - complete(&optimizer_completion); + /* Step 5: Wake up flushers, and kick optimizer again if needed. */ + optimizer_passes++; + wake_up_var_locked(&optimizer_passes, &kprobe_mutex); if (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) kick_kprobe_optimizer(); /*normal kick*/ @@ -708,7 +710,8 @@ static void wait_for_kprobe_optimizer_locked(void) lockdep_assert_held(&kprobe_mutex); while (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) { - init_completion(&optimizer_completion); + unsigned long passes = optimizer_passes; + /* * Set state to OPTIMIZER_ST_FLUSHING and wake up the thread if it's * idle. If it's already kicked, it will see the state change. @@ -717,9 +720,12 @@ static void wait_for_kprobe_optimizer_locked(void) OPTIMIZER_ST_FLUSHING) != OPTIMIZER_ST_FLUSHING) wake_up(&kprobe_optimizer_wait); - mutex_unlock(&kprobe_mutex); - wait_for_completion(&optimizer_completion); - mutex_lock(&kprobe_mutex); + /* + * kprobe_optimizer() holds 'kprobe_mutex' for a whole pass, which + * this drops while sleeping, so a new count means a full pass ran. + */ + wait_var_event_mutex(&optimizer_passes, + optimizer_passes != passes, &kprobe_mutex); } } diff --git a/kernel/power/hibernate.c b/kernel/power/hibernate.c index d2479c69d71a..c13f68ab7f6e 100644 --- a/kernel/power/hibernate.c +++ b/kernel/power/hibernate.c @@ -408,9 +408,18 @@ int hibernation_snapshot(int platform_mode) if (error) goto Close; + error = dpm_prepare(PMSG_FREEZE); + if (error) + goto Complete; + + /* Preallocate image memory before freezing kernel threads and shutting down devices. */ + error = hibernate_preallocate_memory(); + if (error) + goto Complete; + error = freeze_kernel_threads(); if (error) - goto Close; + goto Cleanup; if (hibernation_test(TEST_FREEZER)) { @@ -422,15 +431,6 @@ int hibernation_snapshot(int platform_mode) goto Thaw; } - error = dpm_prepare(PMSG_FREEZE); - if (error) - goto Complete; - - /* Preallocate image memory before shutting down devices. */ - error = hibernate_preallocate_memory(); - if (error) - goto Complete; - console_suspend_all(); pm_restrict_gfp_mask(); @@ -464,10 +464,12 @@ int hibernation_snapshot(int platform_mode) platform_end(platform_mode); return error; - Complete: - dpm_complete(PMSG_RECOVER); Thaw: thaw_kernel_threads(); + Cleanup: + swsusp_free(); + Complete: + dpm_complete(PMSG_RECOVER); goto Close; } diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 0b846a13c628..1fe40de6ebe3 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -5798,7 +5798,7 @@ void sched_tick(void) curr = rq->curr; donor = rq->donor; - psi_account_irqtime(rq, donor, NULL); + psi_account_irqtime(rq, curr, NULL); update_rq_clock(rq); hw_pressure = arch_scale_hw_pressure(cpu_of(rq)); diff --git a/kernel/sched/ext/cid.c b/kernel/sched/ext/cid.c index bc4eee5bb4cb..4b08866d75f6 100644 --- a/kernel/sched/ext/cid.c +++ b/kernel/sched/ext/cid.c @@ -912,30 +912,36 @@ bool scx_cmask_empty(const struct scx_cmask *m) /** * scx_bpf_cid_topo - Copy out per-cid topology info * @cid: cid to look up - * @out__uninit: where to copy the topology info; fully written by this call + * @out: where to copy the topology info + * @out__sz: size of @out, the program's sizeof(struct scx_cid_topo) * @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs * - * Fill @out__uninit with the topology info for @cid. Trigger scx_error() if - * @cid is out of range. If @cid is valid but in the no-topo section, all fields - * are set to -1. All fields are also set to -1 when no cid tables have been - * published yet, which a program may observe while racing the root enable. + * Fill @out with the topology info for @cid. Trigger scx_error() if @cid is out + * of range. If @cid is valid but in the no-topo section, all fields are set to + * -1. All fields are also set to -1 when no cid tables have been published yet, + * which a program may observe while racing the root enable. + * + * The program's struct may be older or newer than the kernel's. The smaller of + * @out__sz and the kernel's size is copied and the rest of @out is set to -1. */ -__bpf_kfunc void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out__uninit, +__bpf_kfunc void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out, size_t out__sz, const struct bpf_prog_aux *aux) { + size_t len = min(out__sz, sizeof(*out)); struct scx_cid_topo *topo; struct scx_sched *sch; + /* the error cases and fields the kernel lacks read as -1 */ + memset(out, 0xff, out__sz); + guard(rcu)(); sch = scx_prog_sched(aux); topo = rcu_dereference(scx_cid_topo); - if (unlikely(!sch) || !cid_valid(sch, cid) || unlikely(!topo)) { - *out__uninit = SCX_CID_TOPO_NEG; + if (unlikely(!sch) || !cid_valid(sch, cid) || unlikely(!topo)) return; - } - *out__uninit = topo[cid]; + memcpy(out, &topo[cid], len); } __bpf_kfunc_end_defs(); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 3219f0da0fe4..e56c3c95018f 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -447,37 +447,45 @@ static void switch_rq_lock(struct rq *from, struct rq *to) DEFINE_STATIC_KEY_FALSE(__scx_is_cid_type); /** - * scx_call_op_set_cpumask - invoke ops.set_cpumask / ops_cid.set_cmask for @task + * scx_fill_cmask_scratch - Build this cpu's arena cmask from @cpumask + * @sch: scx_sched whose scratch to fill + * @cpumask: cpus to translate into cids + * + * The scratch lives in BPF-writable arena memory and its header can't be + * trusted, so it is rewritten from kernel geometry rather than read. Caller + * must hold an rq lock so this cpu is the sole kernel writer for as long as the + * returned address is in use. + */ +static struct scx_cmask *scx_fill_cmask_scratch(struct scx_sched *sch, + const struct cpumask *cpumask) +{ + struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch); + struct scx_cmask_ref ref; + + scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref); + scx_cmask_ref_from_cpumask(&ref, cpumask); + return kern_va; +} + +/** + * scx_call_op_set_cpumask - Invoke the set_cpumask or set_cmask op for @task * @sch: scx_sched being invoked * @rq: rq to update as the currently-locked rq, or NULL * @task: task whose affinity is changing * @cpumask: new cpumask * - * For cid-form schedulers, translate @cpumask to a cmask via the per-cpu - * scratch in cid.c and dispatch through the ops_cid union view. Caller - * must hold @rq's rq lock so this_cpu_ptr is stable across the call. + * For cid-form schedulers, translate @cpumask to a cmask in the per-cpu scratch + * and dispatch through the ops_cid union view. Caller must hold @rq's rq lock. */ static inline void scx_call_op_set_cpumask(struct scx_sched *sch, struct rq *rq, struct task_struct *task, const struct cpumask *cpumask) { - if (scx_is_cid_type()) { - struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch); - struct scx_cmask_ref ref; - - /* - * Build the per-cpu arena cmask from kernel geometry via @ref, - * never reading its BPF-writable header. set_cmask()'s __arena - * argument takes the kernel address and the struct_ops - * trampoline rebases it into BPF's arena pointer form. The rq - * lock makes this cpu the sole kernel writer. - */ - scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref); - scx_cmask_ref_from_cpumask(&ref, cpumask); - SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task, kern_va); - } else { + if (scx_is_cid_type()) + SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task, + scx_fill_cmask_scratch(sch, cpumask)); + else SCX_CALL_OP_TASK(sch, set_cpumask, rq, task, cpumask); - } } enum scx_dsq_iter_flags { @@ -1499,27 +1507,22 @@ static inline bool task_scx_migrating(struct task_struct *p) return p->scx.sticky_cpu >= 0; } -/* - * Call ops.dequeue() if the task is in BPF custody and not migrating. - * Clears %SCX_TASK_IN_CUSTODY when the callback is invoked. - */ -static void call_task_dequeue(struct scx_sched *sch, struct rq *rq, - struct task_struct *p, u64 deq_flags) +/* Must be called under the lock serializing @p's custody transfers. */ +static bool task_leave_custody(struct task_struct *p) { if (!(p->scx.flags & SCX_TASK_IN_CUSTODY) || task_scx_migrating(p)) - return; - - if (SCX_HAS_OP(sch, dequeue)) - SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags); + return false; p->scx.flags &= ~SCX_TASK_IN_CUSTODY; + return true; } static void rq_owned_post_enq(struct scx_sched *sch, struct rq *rq, struct scx_dispatch_q *dsq, struct task_struct *p, u64 enq_flags) { - call_task_dequeue(sch, rq, p, 0); + if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue)) + SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0); /* * Only local inserts get the wakeup treatment below. Rejects kick the @@ -1705,20 +1708,28 @@ static void scx_dispatch_enqueue(struct scx_sched *sch, struct rq *rq, if (is_rq_owned) { rq_owned_post_enq(sch, rq, dsq, p, enq_flags); } else { + bool call_dequeue = false; + /* * Global and bypass DSQs are terminal - the task leaves the - * scheduler's custody, so ops.dequeue() fires here. It can run + * scheduler's custody, so ops.dequeue() fires. It can run * without @p's rq lock (finish_dispatch() passes the dispatch * rq); that's safe because dequeue_task_scx() waits on * SCX_OPSS_DISPATCHING (see the ops_state note above) and so * can't race it. A non-terminal DSQ keeps the task in custody. + * The custody transfer happens under @dsq->lock so that later + * consumers see the flag clear; the callback runs after + * @dsq->lock is dropped because it may lock a DSQ itself. */ if (dsq->id == SCX_DSQ_GLOBAL || dsq->id == SCX_DSQ_BYPASS) - call_task_dequeue(sch, rq, p, 0); + call_dequeue = task_leave_custody(p); else p->scx.flags |= SCX_TASK_IN_CUSTODY; raw_spin_unlock(&dsq->lock); + + if (call_dequeue && SCX_HAS_OP(sch, dequeue)) + SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0); } /* @@ -2141,7 +2152,12 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_ int sticky_cpu = p->scx.sticky_cpu; u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags; - if (enq_flags & ENQUEUE_WAKEUP) + /* + * SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue + * returns. Only the core's wakeup path delivers one. The flags stashed + * for a remote activation may carry the wakeup bit without it. + */ + if (core_enq_flags & ENQUEUE_WAKEUP) rq->scx.flags |= SCX_RQ_IN_WAKEUP; /* @@ -2210,7 +2226,7 @@ retry: /* * A queued task must always be in BPF scheduler's custody. If * SCX_TASK_IN_CUSTODY is clear, finish_dispatch() on another - * CPU has already passed call_task_dequeue() (which clears the + * CPU has already passed task_leave_custody() (which clears the * flag), but has not yet written SCX_OPSS_NONE. That final * store does not require this rq's lock, so retrying with * cpu_relax() is bounded: we will observe NONE (or DISPATCHING, @@ -2258,7 +2274,8 @@ retry: * NONE but the task may still have %SCX_TASK_IN_CUSTODY set until * it is enqueued on the destination. */ - call_task_dequeue(sch, rq, p, deq_flags); + if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue)) + SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags); } static bool dequeue_task_scx(struct rq *rq, struct task_struct *p, int core_deq_flags) @@ -2374,14 +2391,10 @@ static void wakeup_preempt_scx(struct rq *rq, struct task_struct *p, int wake_fl } void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p, - u64 enq_flags, struct scx_dispatch_q *src_dsq, - struct rq *dst_rq) + u64 enq_flags, struct rq *dst_rq) { struct scx_dispatch_q *dst_dsq = scx_resolve_local_dsq(sch, dst_rq, p, &enq_flags); - /* @p is on @dst_rq, an rq-owned @src_dsq is covered by the rq lock */ - if (!dsq_is_rq_owned(src_dsq)) - lockdep_assert_held(&src_dsq->lock); lockdep_assert_rq_held(dst_rq); WARN_ON_ONCE(p->scx.holding_cpu >= 0); @@ -2629,8 +2642,8 @@ static struct rq *move_task_between_dsqs(struct scx_sched *sch, /* @p is going from a non-local DSQ to a local DSQ */ if (src_rq == dst_rq) { scx_task_unlink_from_dsq(p, src_dsq); - scx_move_local_task_to_local_dsq(sch, p, enq_flags, src_dsq, dst_rq); raw_spin_unlock(&src_dsq->lock); + scx_move_local_task_to_local_dsq(sch, p, enq_flags, dst_rq); } else { raw_spin_unlock(&src_dsq->lock); move_remote_task_to_local_dsq(sch, p, enq_flags, src_rq, dst_rq); @@ -2680,8 +2693,8 @@ retry: if (rq == task_rq) { scx_task_unlink_from_dsq(p, dsq); - scx_move_local_task_to_local_dsq(sch, p, enq_flags, dsq, rq); raw_spin_unlock(&dsq->lock); + scx_move_local_task_to_local_dsq(sch, p, enq_flags, rq); return true; } @@ -3629,8 +3642,12 @@ static void set_cpus_allowed_scx(struct task_struct *p, * * Fine-grained memory write control is enforced by BPF making the const * designation pointless. Cast it away when calling the operation. + * + * The cid form receives the initial mask when the task is enabled and + * hears about changes only afterwards, see struct scx_enable_args. */ - if (SCX_HAS_OP(sch, set_cpumask)) + if (SCX_HAS_OP(sch, set_cpumask) && + (!scx_is_cid_type() || scx_get_task_state(p) == SCX_TASK_ENABLED)) scx_call_op_set_cpumask(sch, task_rq(p), p, (struct cpumask *)p->cpus_ptr); } @@ -3939,8 +3956,27 @@ static void __scx_enable_task(struct scx_sched *sch, struct task_struct *p) p->scx.weight = sched_weight_to_cgroup(weight); - if (SCX_HAS_OP(sch, enable)) - SCX_CALL_OP_TASK(sch, enable, rq, p); + if (SCX_HAS_OP(sch, enable)) { + if (scx_is_cid_type()) { + struct scx_cmask *cmask = scx_fill_cmask_scratch(sch, p->cpus_ptr); + struct scx_enable_args args = { + .cmask_arena_addr = scx_kaddr_to_arena(sch, cmask), + }; + + SCX_CALL_CID_OP_TASK(sch, enable, rq, p, &args); + } else { + SCX_CALL_OP_TASK(sch, enable, rq, p); + } + } + + /* + * The initial mask also goes out through set_cmask() so a scheduler can + * track affinity there alone, and before set_weight() so that the mask + * is in place when weight-dependent state is derived, see struct + * scx_enable_args. + */ + if (scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask)) + scx_call_op_set_cpumask(sch, rq, p, p->cpus_ptr); if (SCX_HAS_OP(sch, set_weight)) SCX_CALL_OP_TASK(sch, set_weight, rq, p, p->scx.weight); @@ -4283,9 +4319,10 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p) /* * set_cpus_allowed_scx() is not called while @p is associated with a - * different scheduler class. Keep the BPF scheduler up-to-date. + * different scheduler class. Keep the BPF scheduler up-to-date. The cid + * form gets its mask from scx_enable_task(). */ - if (SCX_HAS_OP(sch, set_cpumask)) + if (!scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask)) scx_call_op_set_cpumask(sch, rq, p, (struct cpumask *)p->cpus_ptr); } @@ -4404,6 +4441,17 @@ static bool local_task_should_reenq(struct rq *rq, struct task_struct *p, return *reenq_flags & SCX_REENQ_ANY; } +/* + * The dispatcher stores the final ops_state after dropping the DSQ lock, so @p + * can be found on a DSQ while still %SCX_OPSS_DISPATCHING. Reenqueueing @p + * before that store lands would have it clobber the new %SCX_OPSS_QUEUED. + */ +void scx_reenq_wait_dispatching(struct task_struct *p) +{ + if (unlikely(atomic_long_read_acquire(&p->scx.ops_state) == SCX_OPSS_DISPATCHING)) + wait_ops_state(p, SCX_OPSS_DISPATCHING); +} + static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags) { LIST_HEAD(tasks); @@ -4447,6 +4495,7 @@ static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags) if (!local_task_should_reenq(rq, p, &reenq_flags, &reason)) continue; + scx_reenq_wait_dispatching(p); scx_dispatch_dequeue(rq, p); if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK)) @@ -4570,6 +4619,7 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag } /* @p is on @dsq, its rq and @dsq are locked */ + scx_reenq_wait_dispatching(p); dispatch_dequeue_locked(p, dsq); raw_spin_unlock(&dsq->lock); @@ -8356,10 +8406,11 @@ static struct bpf_struct_ops bpf_sched_ext_ops = { /* * cid-form cfi stubs. Stubs whose signatures match the cpu-form (param types * identical, only param names differ across structs) are reused. Some need - * fresh stubs, set_cmask due to an argument type difference and the sub-sched - * notifiers because no cpu-form stub exists to reuse. + * fresh stubs, set_cmask and enable due to argument differences and the + * sub-sched notifiers because no cpu-form stub exists to reuse. */ static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx_cmask *cmask__arena) {} +static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {} static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {} static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {} @@ -8380,7 +8431,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = { .update_idle = sched_ext_ops__update_idle, .init_task = sched_ext_ops__init_task, .exit_task = sched_ext_ops__exit_task, - .enable = sched_ext_ops__enable, + .enable = sched_ext_ops_cid__enable, .disable = sched_ext_ops__disable, #ifdef CONFIG_EXT_GROUP_SCHED .cpuctl_init = sched_ext_ops__cgroup_init, @@ -10403,8 +10454,7 @@ __bpf_kfunc const void *scx_bpf_online_cmask(const struct bpf_prog_aux *aux) if (unlikely(!online)) return NULL; - /* BPF rebases by the low 32 bits, like __arena callback args */ - return (void *)((unsigned long)online - sch->arena_kern_base); + return (void *)scx_kaddr_to_arena(sch, online); } /** diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h index ed423bcc26b8..2ff5479334cb 100644 --- a/kernel/sched/ext/inlines.h +++ b/kernel/sched/ext/inlines.h @@ -129,8 +129,10 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq, * scheduler's ops.dispatch() doesn't yield any tasks. */ if (scx_bypass_dsp_enabled(sch) && - scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0)) + scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0)) { + __scx_add_event(sch, SCX_EV_SUB_BYPASS_DISPATCH, 1); return SCX_DSP_LOCAL; + } return SCX_DSP_NONE; } diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h index 3464e0f113c1..1df8f583b0ec 100644 --- a/kernel/sched/ext/internal.h +++ b/kernel/sched/ext/internal.h @@ -250,6 +250,31 @@ struct scx_exit_task_args { bool cancelled; }; +/** + * struct scx_enable_args - Argument container for cid-form ops.enable() + * @cmask_arena_addr: BPF arena address of the cmask of cids the task may run on + * + * @cmask_arena_addr is the task's affinity as it enters the scheduler. + * set_cmask() delivers the same mask right after enable(), before set_weight() + * and the first enqueue, then every affinity change afterwards, and is never + * called before enable(). A scheduler may therefore track affinity in + * set_cmask() alone. + * + * The kernel builds the mask in the scheduler arena from its own geometry, so + * the header is valid regardless of what the scheduler last wrote there. The + * memory is per-cpu scratch reused once the callback returns: copy the bits + * out, don't keep the address. The set_cmask() argument follows the same rules. + * + * The address is a plain value rather than a typed pointer because BTF can't + * mark a struct member as an arena pointer yet and a pointer member would reach + * the program typed as a kernel pointer. Cast it to struct scx_cmask __arena * + * before use. Once arena members can be typed, a typed alias will join this + * field in an anonymous union at the same offset. + */ +struct scx_enable_args { + u64 cmask_arena_addr; +}; + /* argument container for ops.cgroup_init() */ struct scx_cgroup_init_args { /* the weight of the cgroup [1..10000] */ @@ -1037,6 +1062,7 @@ struct sched_ext_ops { * - dispatch -> dispatch (cpu arg is now cid) * - update_idle -> update_idle (cpu arg is now cid) * - set_cpumask -> set_cmask (cmask instead of cpumask) + * - enable -> enable (takes struct scx_enable_args) * - cpu_online -> cid_online * - cpu_offline -> cid_offline * - dump_cpu -> dump_cid @@ -1070,7 +1096,7 @@ struct sched_ext_ops_cid { struct scx_init_task_args *args); void (*exit_task)(struct task_struct *p, struct scx_exit_task_args *args); - void (*enable)(struct task_struct *p); + void (*enable)(struct task_struct *p, struct scx_enable_args *args); void (*disable)(struct task_struct *p); void (*dump)(struct scx_dump_ctx *ctx); void (*dump_cid)(struct scx_dump_ctx *ctx, s32 cid, bool idle); @@ -1533,7 +1559,8 @@ struct scx_sched { * by BUILD_BUG_ON in scx_init()). The anonymous union lets the kernel * access either view of the same storage without function-pointer * casts: use .ops for cpu-form and shared fields, .ops_cid for the - * cid-renamed callbacks (set_cmask, select_cid, cid_online, ...). + * callbacks whose cid-form signature differs (set_cmask, enable, + * select_cid, cid_online, ...). */ union { struct sched_ext_ops ops; @@ -1556,9 +1583,9 @@ struct scx_sched { uintptr_t arena_kern_base; /* - * Per-CPU arena cmask used by scx_call_op_set_cpumask() to hand a cmask - * to ops_cid.set_cmask(). The kernel writes through the stored kern_va - * and passes it to the callback's __arena argument. + * Per-CPU arena cmask the kernel fills from a task's cpumask and hands + * to ops_cid.enable() and ops_cid.set_cmask(). The stored pointers are + * the kernel addresses. */ struct scx_cmask * __percpu *set_cmask_scratch; struct scx_cmask *online_cmask; @@ -1669,6 +1696,19 @@ static inline void *scx_arena_to_kaddr(struct scx_sched *sch, const void *bpf_pt return (void *)(sch->arena_kern_base + (u32)(uintptr_t)bpf_ptr); } +/** + * scx_kaddr_to_arena - Translate a kernel arena address to the BPF form + * @sch: scheduler whose arena hosts @kaddr + * @kaddr: kernel address inside @sch's arena + * + * __arena callback arguments need no translation. Addresses handed to BPF any + * other way, such as struct fields and kfunc return values, go through this. + */ +static inline uintptr_t scx_kaddr_to_arena(struct scx_sched *sch, const void *kaddr) +{ + return (uintptr_t)kaddr - sch->arena_kern_base; +} + enum scx_wake_flags { /* expose select WF_* flags as enums */ SCX_WAKE_FORK = WF_FORK, @@ -2065,8 +2105,7 @@ void scx_dispatch_dequeue(struct rq *rq, struct task_struct *p); void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags, int sticky_cpu); void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p, - u64 enq_flags, struct scx_dispatch_q *src_dsq, - struct rq *dst_rq); + u64 enq_flags, struct rq *dst_rq); bool scx_consume_dispatch_q(struct scx_sched *sch, struct rq *rq, struct scx_dispatch_q *dsq, u64 enq_flags); bool scx_consume_global_dsq(struct scx_sched *sch, struct rq *rq); @@ -2078,6 +2117,7 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags); u64 __scx_bpf_now(struct rq *rq); void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq, u64 reenq_flags, struct rq *locked_rq); +void scx_reenq_wait_dispatching(struct task_struct *p); int __scx_init_task(struct scx_sched *sch, struct task_struct *p, struct cgroup *cgrp, bool fork); void scx_enable_task(struct scx_sched *sch, struct task_struct *p); @@ -2302,9 +2342,9 @@ do { \ } while (0) /* - * Dispatch a task op through the cid-form ops_cid table. Only set_cmask() needs - * this: it takes an arena cmask address instead of a cpumask, so it cannot be - * invoked via its cpu-form set_cpumask() slot. + * Dispatch a task op through the cid-form ops_cid table, for the ops whose + * cid-form signature differs from the cpu-form slot: set_cmask() takes an arena + * cmask instead of a cpumask and enable() takes scx_enable_args. */ #define SCX_CALL_CID_OP_TASK(sch, op, locked_rq, task, args...) \ __SCX_CALL_OP_TASK(sch, ops_cid, op, locked_rq, task, ##args) diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index 34e642a1a403..0472eaf41c7c 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -555,8 +555,8 @@ static void scx_rescue_timerfn(struct timer_list *timer) scx.dsq_list.node); scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq); scx_rescue_admit(rq, p, slice); - scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS, - &rq->scx.rescue.dsq, rq); + scx_move_local_task_to_local_dsq(scx_task_sched(p), p, + SCX_ENQ_IGNORE_CAPS, rq); if (sched_class_above(&ext_sched_class, rq->curr->sched_class)) resched_curr(rq); } else if (p->scx.dsq && rq->scx.rescue.budget > 2 * scx_rescue_quantum_ns) { @@ -572,7 +572,7 @@ static void scx_rescue_timerfn(struct timer_list *timer) scx_task_unlink_from_dsq(p, &rq->scx.local_dsq); scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_HEAD | SCX_ENQ_PREEMPT | SCX_ENQ_IGNORE_CAPS, - &rq->scx.local_dsq, rq); + rq); } out_arm: scx_rescue_timer_arm(rq); @@ -596,8 +596,8 @@ void scx_rescue_flush(struct rq *rq) /* and flush out all pending ones */ list_for_each_entry_safe(p, n, &rq->scx.rescue.dsq.list, scx.dsq_list.node) { scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq); - scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS, - &rq->scx.rescue.dsq, rq); + scx_move_local_task_to_local_dsq(scx_task_sched(p), p, + SCX_ENQ_IGNORE_CAPS, rq); } timer_delete(&rq->scx.rescue.timer); @@ -801,6 +801,7 @@ void scx_reenq_reject(struct rq *rq) if (WARN_ON_ONCE(p->migration_pending)) continue; + scx_reenq_wait_dispatching(p); scx_dispatch_dequeue(rq, p); if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK)) diff --git a/kernel/sched/ext/types.h b/kernel/sched/ext/types.h index 943d8d429a2c..139176cf9fc6 100644 --- a/kernel/sched/ext/types.h +++ b/kernel/sched/ext/types.h @@ -70,6 +70,10 @@ enum scx_consts { * smaller shards if the LLC exceeds the target size. No-topo cids are packed * into their own max-sized shards. * + * New fields are appended, never inserted: scx_bpf_cid_topo() copies this + * struct out sized by the program's own layout, and an older program's copy + * must stay a prefix of the kernel's. + * * @core_cid: first cid of this cid's core (smt-sibling group) * @core_idx: global index of that core, in [0, nr_cores_at_init) * @llc_cid: first cid of this cid's LLC diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 7455a83a6a99..57360f5cdde4 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -1478,7 +1478,7 @@ static inline int get_sched_cache_scale(int mul) return (1 + (tol - 1) * mul); } -static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) +static bool exceed_llc_capacity(struct sched_cache_group *grp, int cpu) { #ifdef CONFIG_NUMA_BALANCING unsigned long llc, footprint; @@ -1497,7 +1497,7 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) * excluded. */ llc = sd->llc_bytes; - footprint = READ_ONCE(mm->sc_stat.footprint); + footprint = READ_ONCE(grp->footprint); /* * Scale the LLC size by 256*llc_aggr_tolerance @@ -1526,7 +1526,7 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) return false; } -static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, +static bool invalid_llc_nr(struct sched_cache_group *grp, struct task_struct *p, int cpu) { int scale; @@ -1542,10 +1542,32 @@ static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, if (scale == INT_MAX) return false; - return !fits_capacity((mm->sc_stat.nr_running_avg * cpu_smt_num_threads), + return !fits_capacity((READ_ONCE(grp->nr_running_avg) * cpu_smt_num_threads), (scale * per_cpu(sd_llc_size, cpu))); } +/* + * A task counts in nr_pref_llc_running while it is queued on its preferred + * LLC (pref_llc_queued) and runnable (!sched_delayed), keeping the counter in + * the runnable domain so alb_break_llc() can compare it with h_nr_runnable. + */ +static bool task_pref_llc_runnable(struct task_struct *p) +{ + return p->pref_llc_queued && !p->se.sched_delayed; +} + +static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) +{ + if (task_pref_llc_runnable(p)) + rq->nr_pref_llc_running++; +} + +static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) +{ + if (task_pref_llc_runnable(p)) + rq->nr_pref_llc_running--; +} + static void account_llc_enqueue(struct rq *rq, struct task_struct *p) { int pref_llc, pref_llc_queued; @@ -1557,7 +1579,6 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) pref_llc_queued = (pref_llc == task_llc(p)); rq->nr_llc_running++; - rq->nr_pref_llc_running += pref_llc_queued; /* * Record whether p is enqueued on its preferred @@ -1575,6 +1596,9 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) */ p->pref_llc_queued = pref_llc_queued; + /* Skipped while delayed; clear_delayed() adds it back on wake. */ + pref_llc_running_inc(rq, p); + sd = rcu_dereference_all(rq->sd); if (sd && (unsigned int)pref_llc < sd->llc_max) sd->llc_counts[pref_llc]++; @@ -1591,7 +1615,12 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p) rq->nr_llc_running--; if (p->pref_llc_queued) { - rq->nr_pref_llc_running--; + /* + * Skipped if still delayed (set_delayed() already removed it); + * clearing pref_llc_queued below also stops clear_delayed() + * from re-adding it. + */ + pref_llc_running_dec(rq, p); /* * Update the status in case * other logic might query @@ -1619,12 +1648,20 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p) } } -void mm_init_sched(struct mm_struct *mm, - struct sched_cache_time __percpu *_pcpu_sched) +int mm_init_sched(struct mm_struct *mm, + struct sched_cache_time __percpu *_pcpu_sched) { + struct sched_cache_group *grp; unsigned long epoch = 0; int i; + grp = kzalloc_obj(*grp); + if (!grp) { + free_percpu(_pcpu_sched); + mm->sched_cache_grp = NULL; + return -ENOMEM; + } + for_each_possible_cpu(i) { struct sched_cache_time *pcpu_sched = per_cpu_ptr(_pcpu_sched, i); struct rq *rq = cpu_rq(i); @@ -1635,18 +1672,141 @@ void mm_init_sched(struct mm_struct *mm, epoch = rq->cpu_epoch; } - raw_spin_lock_init(&mm->sc_stat.lock); - mm->sc_stat.epoch = epoch; - mm->sc_stat.cpu = -1; - mm->sc_stat.next_scan = jiffies; - mm->sc_stat.nr_running_avg = 0; - mm->sc_stat.footprint = 0; + raw_spin_lock_init(&grp->lock); + grp->epoch = epoch; + grp->cpu = -1; + grp->next_scan = jiffies; + grp->nr_running_avg = 0; + grp->footprint = 0; + refcount_set(&grp->refcnt, 1); /* - * The update to mm->sc_stat should not be reordered - * before initialization to mm's other fields, in case + * The update to grp->pcpu_sched should not be reordered + * before initialization to grp's other fields, in case * the readers may get invalid mm_sched_epoch, etc. */ - smp_store_release(&mm->sc_stat.pcpu_sched, _pcpu_sched); + smp_store_release(&grp->pcpu_sched, _pcpu_sched); + /* + * Publish the group last. Not every reader qualifies it by + * grp->pcpu_sched - can_migrate_llc_task() only checks that the + * pointer is non-NULL before reading grp->footprint and + * grp->nr_running_avg - so a reachable group must already be + * fully initialized. + */ + smp_store_release(&mm->sched_cache_grp, grp); + return 0; +} + +static void sched_cache_group_free_rcu(struct rcu_head *rcu) +{ + struct sched_cache_group *grp = + container_of(rcu, struct sched_cache_group, rcu); + + free_percpu(grp->pcpu_sched); + kfree(grp); +} + +static void sched_cache_group_put(struct sched_cache_group *grp) +{ + if (!grp || !refcount_dec_and_test(&grp->refcnt)) + return; + + call_rcu(&grp->rcu, sched_cache_group_free_rcu); +} + +DEFINE_FREE(sched_cache_group_put, struct sched_cache_group *, + sched_cache_group_put(_T)); + +#define rcu_deref_sched_cache_grp(tsk) \ + rcu_dereference_check((tsk)->sched_cache_grp, (tsk) == current) + +static struct sched_cache_group *sched_cache_replace_grp(struct task_struct *p, + struct sched_cache_group *new) +{ + struct sched_cache_group *old; + + old = rcu_deref_sched_cache_grp(p); + rcu_assign_pointer(p->sched_cache_grp, new); + + return old; +} + +struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp) +{ + /* + * refcount_inc_not_zero() is the acquire primitive for lockless + * (RCU) lookups; plain refcount_inc() would scribble the count if + * it already reached zero. Return NULL in that case. + */ + if (grp && !refcount_inc_not_zero(&grp->refcnt)) + grp = NULL; + + return grp; +} + +struct sched_cache_group *task_cache_group_get(struct task_struct *p) +{ + guard(rcu)(); + return sched_cache_group_get(rcu_dereference(p->sched_cache_grp)); +} + +void sched_cache_fork(struct task_struct *p) +{ + /* + * The child takes its own reference on the mm's cache group, separate + * from the reference held by the mm. @p is not yet visible to readers, + * so a plain initializing store is enough. + */ + RCU_INIT_POINTER(p->sched_cache_grp, + sched_cache_group_get(p->mm->sched_cache_grp)); +} + +void sched_cache_fork_cleanup(struct task_struct *p) +{ + /* + * A fork that fails after sched_cache_fork() never reaches exit_mm(), + * so drop the reference here. @p never became visible, so there are no + * concurrent readers and the reference we hold keeps the group alive. + */ + sched_cache_group_put(rcu_access_pointer(p->sched_cache_grp)); + RCU_INIT_POINTER(p->sched_cache_grp, NULL); +} + +void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm) +{ + struct sched_cache_group *old; + + /* + * Acquire the new reference before publishing the pointer, then drop + * the old one. @p is current and the only writer of its own pointer. + */ + old = sched_cache_replace_grp(p, sched_cache_group_get(mm->sched_cache_grp)); + sched_cache_group_put(old); +} + +void sched_cache_exit_mm(struct task_struct *p) +{ + struct sched_cache_group *grp = sched_cache_replace_grp(p, NULL); + +#ifdef CONFIG_NUMA_BALANCING + /* + * Subtract this task's footprint from the group before dropping the + * reference, so the group footprint converges as its threads exit. + * Unlocked for performance; clamp to avoid underflow. + */ + if (grp && p->total_numa_faults) { + unsigned long fp = READ_ONCE(grp->footprint); + unsigned long sub = min(fp, p->total_numa_faults); + + WRITE_ONCE(grp->footprint, fp - sub); + } +#endif + sched_cache_group_put(grp); +} + +void mm_destroy_sched(struct mm_struct *mm) +{ + sched_cache_group_put(mm->sched_cache_grp); + mm->sched_cache_grp = NULL; } /* because why would C be fully specified */ @@ -1697,14 +1857,14 @@ static unsigned long fraction_mm_sched(struct rq *rq, return div64_u64(NICE_0_LOAD * pcpu_sched->runtime, rq->cpu_runtime + 1); } -static int get_pref_llc(struct task_struct *p, struct mm_struct *mm) +static int get_pref_llc(struct task_struct *p, struct sched_cache_group *grp) { int mm_sched_llc = -1, mm_sched_cpu; - if (!mm) + if (!grp) return -1; - mm_sched_cpu = READ_ONCE(mm->sc_stat.cpu); + mm_sched_cpu = READ_ONCE(grp->cpu); if (mm_sched_cpu != -1) { mm_sched_llc = llc_id(mm_sched_cpu); @@ -1734,8 +1894,8 @@ static unsigned int task_running_on_cpu(int cpu, struct task_struct *p); static inline void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) { + struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp); struct sched_cache_time *pcpu_sched; - struct mm_struct *mm = p->mm; int mm_sched_llc = -1; unsigned long epoch; @@ -1746,12 +1906,18 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) return; /* * init_task, kthreads and user thread created - * by user_mode_thread() don't have mm. + * by user_mode_thread() don't have a cache group. + * In theory a kernel thread does not have any valid + * cache group, because sched_cache_fork() is not + * invoked for a kernel thread - !grp should gate the + * kernel thread. Use the PF_KTHREAD check explicitly + * here for safety reasons, to guard against future + * modifications and to pair with task_tick_cache(). */ - if (!mm || !mm->sc_stat.pcpu_sched) + if (p->flags & PF_KTHREAD || !grp || !grp->pcpu_sched) return; - pcpu_sched = per_cpu_ptr(mm->sc_stat.pcpu_sched, cpu_of(rq)); + pcpu_sched = per_cpu_ptr(grp->pcpu_sched, cpu_of(rq)); scoped_guard (raw_spinlock, &rq->cpu_epoch_lock) { __update_mm_sched(rq, pcpu_sched); @@ -1764,14 +1930,14 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) * If this process hasn't hit task_cache_work() for a while invalidate * its preferred state. */ - if ((long)(epoch - READ_ONCE(mm->sc_stat.epoch)) > llc_epoch_affinity_timeout || - invalid_llc_nr(mm, p, cpu_of(rq)) || - exceed_llc_capacity(mm, cpu_of(rq))) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if ((long)(epoch - READ_ONCE(grp->epoch)) > llc_epoch_affinity_timeout || + invalid_llc_nr(grp, p, cpu_of(rq)) || + exceed_llc_capacity(grp, cpu_of(rq))) { + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); } - mm_sched_llc = get_pref_llc(p, mm); + mm_sched_llc = get_pref_llc(p, grp); /* task not on rq accounted later in account_entity_enqueue() */ if (task_running_on_cpu(rq->cpu, p) && @@ -1784,31 +1950,32 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) static void task_tick_cache(struct rq *rq, struct task_struct *p) { + struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp); struct callback_head *work = &p->cache_work; - struct mm_struct *mm = p->mm; unsigned long epoch; if (!sched_cache_enabled()) return; - if (!mm || p->flags & PF_KTHREAD || - !mm->sc_stat.pcpu_sched) + if (!grp || p->flags & PF_KTHREAD || + !grp->pcpu_sched) return; epoch = rq->cpu_epoch; /* avoid moving backwards */ - if (time_after_eq(mm->sc_stat.epoch, epoch)) + if (time_after_eq(grp->epoch, epoch)) return; - guard(raw_spinlock)(&mm->sc_stat.lock); + guard(raw_spinlock)(&grp->lock); if (work->next == work) { task_work_add(p, work, TWA_RESUME); - WRITE_ONCE(mm->sc_stat.epoch, epoch); + WRITE_ONCE(grp->epoch, epoch); } } -static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p) +static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p, + struct sched_cache_group *grp) { #ifdef CONFIG_NUMA_BALANCING int cpu, curr_cpu, nid, pref_nid; @@ -1816,7 +1983,7 @@ static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p) if (!static_branch_likely(&sched_numa_balancing)) goto out; - cpu = READ_ONCE(p->mm->sc_stat.cpu); + cpu = READ_ONCE(grp->cpu); if (cpu != -1) nid = cpu_to_node(cpu); curr_cpu = task_cpu(p); @@ -1873,13 +2040,13 @@ static inline void update_avg_scale(u64 *avg, u64 sample) static void task_cache_work(struct callback_head *work) { + struct sched_cache_group *grp __free(sched_cache_group_put) = NULL; + cpumask_var_t cpus __free(free_cpumask_var) = CPUMASK_VAR_NULL; int cpu, m_a_cpu = -1, nr_running = 0, curr_cpu; unsigned long next_scan, now = jiffies; struct task_struct *p = current, *cur; unsigned long curr_m_a_occ = 0; - struct mm_struct *mm = p->mm; unsigned long m_a_occ = 0; - cpumask_var_t cpus; WARN_ON_ONCE(work != &p->cache_work); @@ -1888,21 +2055,30 @@ static void task_cache_work(struct callback_head *work) if (p->flags & PF_EXITING) return; - next_scan = READ_ONCE(mm->sc_stat.next_scan); + /* + * A reference makes sure grp is not released by others. The rcu + * lock can not be held till after zalloc_cpumask_var() below, + * because the latter might sleep. + */ + grp = task_cache_group_get(p); + if (!grp) + return; + + next_scan = READ_ONCE(grp->next_scan); if (time_before(now, next_scan)) return; /* only 1 thread is allowed to scan */ - if (!try_cmpxchg(&mm->sc_stat.next_scan, &next_scan, + if (!try_cmpxchg(&grp->next_scan, &next_scan, now + max_t(unsigned long, READ_ONCE(llc_epoch_period), 1))) return; curr_cpu = task_cpu(p); - if (invalid_llc_nr(mm, p, curr_cpu) || - exceed_llc_capacity(mm, curr_cpu)) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if (invalid_llc_nr(grp, p, curr_cpu) || + exceed_llc_capacity(grp, curr_cpu)) { + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); return; } @@ -1913,7 +2089,7 @@ static void task_cache_work(struct callback_head *work) scoped_guard (cpus_read_lock) { guard(rcu)(); - get_scan_cpumasks(cpus, p); + get_scan_cpumasks(cpus, p, grp); for_each_cpu(cpu, cpus) { /* XXX sched_cluster_active */ @@ -1926,16 +2102,20 @@ static void task_cache_work(struct callback_head *work) for_each_cpu(i, sched_domain_span(sd)) { occ = fraction_mm_sched(cpu_rq(i), - per_cpu_ptr(mm->sc_stat.pcpu_sched, i)); + per_cpu_ptr(grp->pcpu_sched, i)); a_occ += occ; if (occ > m_occ) { m_occ = occ; m_cpu = i; } + /* + * rcu_access_pointer() is used because the + * pointer is only compared, never dereferenced. + */ cur = rcu_dereference_all(cpu_rq(i)->curr); if (cur && !(cur->flags & (PF_EXITING | PF_KTHREAD)) && - cur->mm == mm) + rcu_access_pointer(cur->sched_cache_grp) == grp) nr_running++; } @@ -1959,7 +2139,7 @@ static void task_cache_work(struct callback_head *work) m_a_cpu = m_cpu; } - if (llc_id(cpu) == llc_id(READ_ONCE(mm->sc_stat.cpu))) + if (llc_id(cpu) == llc_id(READ_ONCE(grp->cpu))) curr_m_a_occ = a_occ; cpumask_andnot(cpus, cpus, sched_domain_span(sd)); @@ -1968,7 +2148,7 @@ static void task_cache_work(struct callback_head *work) if (m_a_occ > (2 * curr_m_a_occ)) { /* - * Avoid switching sc_stat.cpu too fast. + * Avoid switching sched_cache_grp->cpu too fast. * The reason to choose 2X is because: * 1. It is better to keep the preferred LLC stable, * rather than changing it frequently and cause migrations @@ -1977,11 +2157,10 @@ static void task_cache_work(struct callback_head *work) * 3. 2X is chosen based on test results, as it delivers * the optimal performance gain so far. */ - WRITE_ONCE(mm->sc_stat.cpu, m_a_cpu); + WRITE_ONCE(grp->cpu, m_a_cpu); } - update_avg_scale(&mm->sc_stat.nr_running_avg, nr_running); - free_cpumask_var(cpus); + update_avg_scale(&grp->nr_running_avg, nr_running); } void init_sched_mm(struct task_struct *p) @@ -1991,10 +2170,18 @@ void init_sched_mm(struct task_struct *p) init_task_work(work, task_cache_work); work->next = work; /* + * dup_task_struct() copies the parent's task_struct, including its + * sched_cache_grp, for which the child holds no reference. Clear it + * here - before copy_mm() runs - so the child never carries a + * borrowed pointer that the fork error path would put. + */ + RCU_INIT_POINTER(p->sched_cache_grp, NULL); + /* * Reset new task's preference to avoid * polluting account_llc_enqueue(). */ p->preferred_llc = -1; + p->pref_llc_queued = 0; } #else /* CONFIG_SCHED_CACHE */ @@ -2016,6 +2203,10 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) {} static void account_llc_dequeue(struct rq *rq, struct task_struct *p) {} +static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) {} + +static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) {} + #endif /* CONFIG_SCHED_CACHE */ /* @@ -3692,6 +3883,7 @@ static int preferred_group_nid(struct task_struct *p, int nid) static void task_numa_placement(struct task_struct *p) __context_unsafe(/* conditional locking */) { + struct sched_cache_group __maybe_unused *grp; int seq, nid, max_nid = NUMA_NO_NODE; unsigned long max_faults = 0; unsigned long fault_types[2] = { 0, 0 }; @@ -3784,19 +3976,24 @@ static void task_numa_placement(struct task_struct *p) * heuristic and occasional lost updates are tolerable. * * If a task exits, its corresponding footprint must - * be subtracted from the mm->sc_stat.footprint, otherwise - * the mm->sc_stat.footprint will not converge: - * the exiting thread's footprint remains unchanged/undecayed - * in mm->sc_stat.footprint. See exit_mm(). + * be subtracted from p->sched_cache_grp->footprint, + * otherwise the footprint will not converge: the + * exiting thread's footprint remains unchanged/undecayed. + * See exit_mm(). * * Lost updates and unsynchronized subtraction * in exit_mm() can cause footprint + diff to * go negative. Clamp to zero to prevent the * unsigned footprint from wrapping. */ - new_fp = (long)READ_ONCE(p->mm->sc_stat.footprint) + diff; - WRITE_ONCE(p->mm->sc_stat.footprint, - max(new_fp, 0L)); + scoped_guard(rcu) { + grp = rcu_dereference(p->sched_cache_grp); + + if (grp) { + new_fp = (long)READ_ONCE(grp->footprint) + diff; + WRITE_ONCE(grp->footprint, max(new_fp, 0L)); + } + } #endif } @@ -6390,15 +6587,27 @@ static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq); static void set_delayed(struct sched_entity *se) { - se->sched_delayed = 1; - /* * Delayed se of cfs_rq have no tasks queued on them. * Do not adjust h_nr_runnable since __dequeue_task() * will account it for blocked tasks. + * + * This check can be removed because when flat pick + * patches get merged as only task can get delayed, + * same for clear_delayed(). */ - if (!entity_is_task(se)) + if (!entity_is_task(se)) { + se->sched_delayed = 1; return; + } + + /* + * Drop a task leaving the runnable set. + * Needs to be called before sched_delayed is set. + * clear_delayed() mirrors this after clearing the flag. + */ + pref_llc_running_dec(rq_of(cfs_rq_of(se)), task_of(se)); + se->sched_delayed = 1; for_each_sched_entity(se) { struct cfs_rq *cfs_rq = cfs_rq_of(se); @@ -6420,6 +6629,13 @@ static void clear_delayed(struct sched_entity *se) if (!entity_is_task(se)) return; + /* + * Re-add on wake, after sched_delayed is cleared. On a final delayed + * dequeue account_llc_dequeue() already cleared pref_llc_queued, so + * this does nothing. + */ + pref_llc_running_inc(rq_of(cfs_rq_of(se)), task_of(se)); + for_each_sched_entity(se) { struct cfs_rq *cfs_rq = cfs_rq_of(se); @@ -10395,6 +10611,7 @@ enum migration_type { #define LBF_SOME_PINNED 0x08 #define LBF_ACTIVE_LB 0x10 #define LBF_LLC_PINNED 0x20 +#define LBF_ACTIVE_LB_LLC 0x40 struct lb_env { struct sched_domain *sd; @@ -10724,7 +10941,7 @@ static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct static enum llc_mig can_migrate_llc_task(struct lb_env *env, struct task_struct *p) { - struct mm_struct *mm; + struct sched_cache_group *grp; bool to_pref; int cpu, src_cpu, dst_cpu; @@ -10733,19 +10950,19 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env, src_cpu = env->src_cpu; dst_cpu = env->dst_cpu; - mm = p->mm; - if (!mm) + grp = rcu_dereference_all(p->sched_cache_grp); + if (!grp) return mig_unrestricted; - cpu = READ_ONCE(mm->sc_stat.cpu); + cpu = READ_ONCE(grp->cpu); if (cpu < 0 || cpus_share_cache(src_cpu, dst_cpu)) return mig_unrestricted; /* skip cache aware load balance for too many threads */ - if (invalid_llc_nr(mm, p, dst_cpu) || - exceed_llc_capacity(mm, dst_cpu)) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if (invalid_llc_nr(grp, p, dst_cpu) || + exceed_llc_capacity(grp, dst_cpu)) { + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); return mig_unrestricted; } @@ -10814,6 +11031,21 @@ alb_break_llc(struct lb_env *env) } /* + * Returns true if p's preferred LLC does not match the destination CPU + * under migrate_llc_task semantics. Passive LB passes migrate_llc_task + * in env->migration_type, while active LB carries LBF_ACTIVE_LB_LLC in + * env->flags to avoid overwriting env->migration_type. + */ +static inline bool +migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env) +{ + return sched_cache_enabled() && + (env->migration_type == migrate_llc_task || + env->flags & LBF_ACTIVE_LB_LLC) && + READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu); +} + +/* * Check if migrating task p from env->src_cpu to * env->dst_cpu breaks LLC localiy. */ @@ -10841,8 +11073,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env) * run on env->dst_cpu, skip the tasks do not prefer * env->dst_cpu, and find the one that prefers. */ - if (env->migration_type == migrate_llc_task && - READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu)) + if (migrate_llc_task_wrong_dst(p, env)) return true; if (can_migrate_llc_task(env, p) != mig_forbid) @@ -10865,6 +11096,12 @@ alb_break_llc(struct lb_env *env) } static inline bool +migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env) +{ + return false; +} + +static inline bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env) { return false; @@ -10963,7 +11200,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env) * 4) too many balance attempts have failed. */ if (env->flags & LBF_ACTIVE_LB) - return 1; + return !migrate_llc_task_wrong_dst(p, env); degrades = migrate_degrades_locality(p, env); if (!degrades) { @@ -13362,6 +13599,20 @@ static int need_active_balance(struct lb_env *env) } static int active_load_balance_cpu_stop(void *data); +static int active_load_balance_llc_cpu_stop(void *data); + +/* + * migration_type is checked elsewhere to decide migration policy, so + * it shouldn't be repurposed just to flag an LLC-directed active + * balance across the stopper. Pick the callback here instead. + */ +static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env) +{ + if (env->migration_type == migrate_llc_task) + return active_load_balance_llc_cpu_stop; + + return active_load_balance_cpu_stop; +} static int should_we_balance(struct lb_env *env) { @@ -13707,7 +13958,7 @@ more_balance: } if (active_balance) { stop_one_cpu_nowait(cpu_of(busiest), - active_load_balance_cpu_stop, busiest, + alb_stop_fn(&env), busiest, &busiest->active_balance_work); } preempt_enable(); @@ -13812,7 +14063,7 @@ update_next_balance(struct sched_domain *sd, unsigned long *next_balance) * least 1 task to be running on each physical CPU where possible, and * avoids physical / logical imbalances. */ -static int active_load_balance_cpu_stop(void *data) +static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags) { struct rq *busiest_rq = data; int busiest_cpu = cpu_of(busiest_rq); @@ -13862,7 +14113,7 @@ static int active_load_balance_cpu_stop(void *data) .src_cpu = busiest_rq->cpu, .src_rq = busiest_rq, .idle = CPU_IDLE, - .flags = LBF_ACTIVE_LB, + .flags = LBF_ACTIVE_LB | lb_flags, }; schedstat_inc(sd->alb_count); @@ -13890,6 +14141,16 @@ out_unlock: return 0; } +static int active_load_balance_cpu_stop(void *data) +{ + return __active_load_balance_cpu_stop(data, 0); +} + +static int active_load_balance_llc_cpu_stop(void *data) +{ + return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC); +} + /* * Scale the max sched_balance_rq interval with the number of CPUs in the system. * This trades load-balance latency on larger machines for less cross talk. diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c index 0248227d983a..3dab0253976f 100644 --- a/kernel/sched/topology.c +++ b/kernel/sched/topology.c @@ -985,8 +985,8 @@ void sched_cache_active_set(void) } /* - * Update the bottom sched_domain's llc_bytes for @cpu and all its - * LLC siblings. Called from cacheinfo_cpu_online() or + * Update the bottom sched_domain's llc_bytes for @cpus sharing a physical + * LLC. Called from cacheinfo_cpu_online() or * cacheinfo_cpu_pre_down() with cpu hotplug lock held. * * Note: get_effective_llc_bytes() returns 0 on PowerPC. @@ -996,17 +996,13 @@ void sched_cache_active_set(void) * and does not populates the per-CPU struct cpu_cacheinfo array * that get_cpu_cacheinfo_llc() reads. */ -void sched_update_llc_bytes(unsigned int cpu) +void sched_update_llc_bytes(const struct cpumask *cpus) { struct sched_domain *sd, *sdp; unsigned int i; sched_domains_mutex_lock(); - sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, cpu)); - if (!sdp) - goto unlock; - /* * ci->shared_cpu_map is built incrementally as CPUs come * online, so the first CPU in an LLC initially sees @@ -1014,14 +1010,22 @@ void sched_update_llc_bytes(unsigned int cpu) * get_effective_llc_bytes(). Re-evaluating every LLC * sibling on each online event corrects this once the full * shared_cpu_map is known. + * + * The departing CPU's domains have already been detached when + * cacheinfo removes it. Use the surviving cache siblings instead. + * They may belong to different cpuset partitions, so use each CPU's + * own LLC domain to scale its share of the physical cache. */ - for_each_cpu(i, sched_domain_span(sdp)) { + for_each_cpu(i, cpus) { + sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, i)); + if (!sdp) + continue; + sd = rcu_dereference_sched_domain(cpu_rq(i)->sd); if (sd) sd->llc_bytes = get_effective_llc_bytes(i, sdp); } -unlock: sched_domains_mutex_unlock(); } diff --git a/kernel/trace/fprobe.c b/kernel/trace/fprobe.c index 1e9b00997ff2..9f2d98181779 100644 --- a/kernel/trace/fprobe.c +++ b/kernel/trace/fprobe.c @@ -171,6 +171,11 @@ static inline bool write_fprobe_header(unsigned long *stack, static inline void read_fprobe_header(unsigned long *stack, struct fprobe **fp, unsigned int *size_words) { + if (!*stack) { + *fp = NULL; + *size_words = 0; + return; + } *fp = arch_decode_fprobe_header_fp(*stack); *size_words = arch_decode_fprobe_header_size(*stack); } @@ -203,6 +208,12 @@ static inline void read_fprobe_header(unsigned long *stack, { struct __fprobe_header *fph = (struct __fprobe_header *)stack; + if (!*stack) { + *fp = NULL; + *size_words = 0; + return; + } + *fp = fph->fp; *size_words = fph->size_words; } @@ -635,6 +646,10 @@ static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops } } + /* Terminate the list, fgraph_reserve_data() does not clear it. */ + if (used && used < reserved_words) + fgraph_data[used] = 0; + /* If any exit_handler is set, data must be used. */ return used != 0; } diff --git a/kernel/workqueue.c b/kernel/workqueue.c index 1ae3732a2c51..959525393739 100644 --- a/kernel/workqueue.c +++ b/kernel/workqueue.c @@ -3918,7 +3918,7 @@ static void check_flush_dependency(struct workqueue_struct *target_wq, WARN_ONCE(current->flags & PF_MEMALLOC, "workqueue: PF_MEMALLOC task %d(%s) is flushing !WQ_MEM_RECLAIM %s:%ps", current->pid, current->comm, target_wq->name, target_func); - WARN_ONCE(worker && ((worker->current_pwq->wq->flags & + WARN_ONCE(worker && worker->current_pwq && ((worker->current_pwq->wq->flags & (WQ_MEM_RECLAIM | __WQ_LEGACY)) == WQ_MEM_RECLAIM), "workqueue: WQ_MEM_RECLAIM %s:%ps is flushing !WQ_MEM_RECLAIM %s:%ps", worker->current_pwq->wq->name, worker->current_func, |
