diff options
| author | Mark Brown <broonie@kernel.org> | 2026-10-01 15:32:41 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-10-01 15:32:41 +0100 |
| commit | f95293010b95d389d61c70411d44744f42ae9e7f (patch) | |
| tree | 5f92e145b929fd1475c4a888f4746685447bd167 /tools/testing/selftests | |
| parent | 5425865a2d29a8f124d5c49ff647942857e2a5b4 (diff) | |
| parent | 6a119bf5e64d3a6fb0f643e41afa6e79cdb2a553 (diff) | |
| download | linux-next-f95293010b95d389d61c70411d44744f42ae9e7f.tar.gz linux-next-f95293010b95d389d61c70411d44744f42ae9e7f.zip | |
Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git
Diffstat (limited to 'tools/testing/selftests')
27 files changed, 3589 insertions, 106 deletions
diff --git a/tools/testing/selftests/sched_ext/.gitignore b/tools/testing/selftests/sched_ext/.gitignore index ae5491a114c0..54a1fd2af713 100644 --- a/tools/testing/selftests/sched_ext/.gitignore +++ b/tools/testing/selftests/sched_ext/.gitignore @@ -4,3 +4,7 @@ !Makefile !.gitignore !config +!test_modules/ +!test_modules/scx_enq_blocked_test.c +!test_modules/Makefile +test_modules/*.mod.c diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile index 4e06d0baaeec..c5d3a2eaea7d 100644 --- a/tools/testing/selftests/sched_ext/Makefile +++ b/tools/testing/selftests/sched_ext/Makefile @@ -5,10 +5,12 @@ include ../../../scripts/Makefile.arch include ../../../scripts/Makefile.include TEST_GEN_PROGS := runner +TEST_GEN_MODS_DIR := test_modules # override lib.mk's default rules OVERRIDE_TARGETS := 1 include ../lib.mk +include ../cgroup/lib/libcgroup.mk CURDIR := $(abspath .) REPOROOT := $(abspath ../../../..) @@ -53,7 +55,7 @@ ifneq ($(wildcard $(GENHDR)),) GENFLAGS := -DHAVE_GENHDR endif -CFLAGS += -g -O2 -rdynamic -pthread -Wall -Werror $(GENFLAGS) \ +CFLAGS += -g -O2 -pthread -Wall -Werror $(GENFLAGS) \ -I$(INCLUDE_DIR) -I$(GENDIR) -I$(LIBDIR) \ -I$(TOOLSINCDIR) -I$(APIDIR) -I$(CURDIR)/include -I$(SCXTOOLSINCDIR) @@ -62,7 +64,7 @@ ifneq ($(LLVM),) CFLAGS += -Wno-unused-command-line-argument endif -LDFLAGS = -lelf -lz -lpthread -lzstd +LDFLAGS += -lelf -lz -lpthread -lzstd IS_LITTLE_ENDIAN = $(shell $(CC) -dM -E - </dev/null | \ grep 'define __BYTE_ORDER__ __ORDER_LITTLE_ENDIAN__') @@ -153,7 +155,7 @@ $(INCLUDE_DIR)/%.bpf.skel.h: $(SCXOBJ_DIR)/%.bpf.o $(INCLUDE_DIR)/vmlinux.h $(BP override define CLEAN rm -rf $(OUTPUT_DIR) - rm -f $(TEST_GEN_PROGS) + rm -f $(TEST_GEN_PROGS) $(EXTRA_CLEAN) endef # Every testcase takes all of the BPF progs are dependencies by default. This @@ -162,9 +164,12 @@ endef all_test_bpfprogs := $(foreach prog,$(wildcard *.bpf.c),$(INCLUDE_DIR)/$(patsubst %.c,%.skel.h,$(prog))) auto-test-targets := \ + cgroup_nr_cpus \ create_dsq \ dequeue \ dequeue_iter \ + dequeue_remote \ + enq_blocked \ enq_last_no_enq_fails \ ddsp_bogus_dsq_fail \ ddsp_vtimelocal_fail \ @@ -174,6 +179,7 @@ auto-test-targets := \ exit \ hotplug \ init_enable_count \ + kick \ maximal \ maybe_null \ minimal \ @@ -213,7 +219,7 @@ $(testcase-targets): $(SCXOBJ_DIR)/%.o: %.c $(SCXOBJ_DIR)/runner.o $(all_test_bp $(SCXOBJ_DIR)/util.o: util.c | $(SCXOBJ_DIR) $(CC) $(CFLAGS) -c $< -o $@ -$(OUTPUT)/runner: $(SCXOBJ_DIR)/runner.o $(SCXOBJ_DIR)/util.o $(BPFOBJ) $(testcase-targets) +$(OUTPUT)/runner: $(SCXOBJ_DIR)/runner.o $(SCXOBJ_DIR)/util.o $(BPFOBJ) $(LIBCGROUP_O) $(testcase-targets) @echo "$(testcase-targets)" $(CC) $(CFLAGS) -o $@ $^ $(LDFLAGS) diff --git a/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c b/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c index 9dd72d0da29b..f14d7e5bef9c 100644 --- a/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c +++ b/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c @@ -147,23 +147,41 @@ void BPF_STRUCT_OPS(allowed_cpus_exit, struct scx_exit_info *ei) } struct task_cpu_arg { - pid_t pid; + u64 pid; + s64 custom_cpu; }; SEC("syscall") int select_cpu_from_user(struct task_cpu_arg *input) { struct task_struct *p; - int cpu; + struct bpf_cpumask *mask; + s32 cpu; p = bpf_task_from_pid(input->pid); if (!p) return -EINVAL; + mask = bpf_cpumask_create(); + if (!mask) { + bpf_task_release(p); + return -ENOMEM; + } + + /* A negative custom_cpu leaves the custom mask empty. */ + if (input->custom_cpu >= 0) + bpf_cpumask_set_cpu(input->custom_cpu, mask); + bpf_rcu_read_lock(); - cpu = scx_bpf_select_cpu_and(p, bpf_get_smp_processor_id(), 0, p->cpus_ptr, 0); + cpu = scx_bpf_select_cpu_and(p, bpf_get_smp_processor_id(), 0, + cast_mask(mask), 0); + if (cpu >= 0 && + (!bpf_cpumask_test_cpu(cpu, cast_mask(mask)) || + !bpf_cpumask_test_cpu(cpu, &p->cpus_mask))) + cpu = -ERANGE; bpf_rcu_read_unlock(); + bpf_cpumask_release(mask); bpf_task_release(p); return cpu; diff --git a/tools/testing/selftests/sched_ext/allowed_cpus.c b/tools/testing/selftests/sched_ext/allowed_cpus.c index 093f285ab4ba..773699d120ea 100644 --- a/tools/testing/selftests/sched_ext/allowed_cpus.c +++ b/tools/testing/selftests/sched_ext/allowed_cpus.c @@ -2,7 +2,10 @@ /* * Copyright (c) 2025 Andrea Righi <arighi@nvidia.com> */ +#define _GNU_SOURCE #include <bpf/bpf.h> +#include <limits.h> +#include <sched.h> #include <scx/common.h> #include <sys/wait.h> #include <unistd.h> @@ -23,17 +26,19 @@ static enum scx_test_status setup(void **ctx) return SCX_TEST_PASS; } -static int test_select_cpu_from_user(const struct allowed_cpus *skel) +static int test_select_cpu_from_user(const struct allowed_cpus *skel, + const char *name, int custom_cpu, + bool expect_busy) { int fd, ret; - __u64 args[1]; + __s32 cpu; + __u64 args[] = { getpid(), (__u64)(__s64)custom_cpu }; LIBBPF_OPTS(bpf_test_run_opts, attr, .ctx_in = args, .ctx_size_in = sizeof(args), ); - args[0] = getpid(); fd = bpf_program__fd(skel->progs.select_cpu_from_user); if (fd < 0) return fd; @@ -42,29 +47,123 @@ static int test_select_cpu_from_user(const struct allowed_cpus *skel) if (ret < 0) return ret; - fprintf(stderr, "%s: CPU %d\n", __func__, attr.retval); + /* test_run returns the signed BPF result through an unsigned field. */ + cpu = (__s32)attr.retval; + if ((expect_busy && cpu != -EBUSY) || + (!expect_busy && cpu != -EBUSY && cpu != custom_cpu)) { + SCX_ERR("%s: unexpected CPU selection result %d", name, cpu); + return -EINVAL; + } return 0; } +/* Grow until the mask covers the kernel's CPU range, including offline CPUs. */ +static int alloc_affinity(cpu_set_t **mask, size_t *size) +{ + int nr_cpus = CPU_SETSIZE; + cpu_set_t *cpus; + int err; + + for (;;) { + *size = CPU_ALLOC_SIZE(nr_cpus); + cpus = CPU_ALLOC(nr_cpus); + if (!cpus) + return -ENOMEM; + CPU_ZERO_S(*size, cpus); + if (!sched_getaffinity(0, *size, cpus)) { + *mask = cpus; + return nr_cpus; + } + err = errno; + CPU_FREE(cpus); + if (err != EINVAL) + return -err; + if (nr_cpus > INT_MAX / 2) + return -EOVERFLOW; + nr_cpus *= 2; + } +} + static enum scx_test_status run(void *ctx) { struct allowed_cpus *skel = ctx; - struct bpf_link *link; + enum scx_test_status status = SCX_TEST_FAIL; + cpu_set_t *original = NULL, *pinned = NULL; + bool affinity_changed = false; + size_t size; + int first = -1, second = -1, cpu, nr_cpus; + struct bpf_link *link = NULL; + + nr_cpus = alloc_affinity(&original, &size); + if (nr_cpus < 0) { + SCX_ERR("Failed to get affinity (%d)", -nr_cpus); + goto out; + } + pinned = CPU_ALLOC(nr_cpus); + if (!pinned) { + SCX_ERR("Failed to allocate affinity mask"); + goto out; + } + for (cpu = 0; cpu < nr_cpus; cpu++) { + if (!CPU_ISSET_S(cpu, size, original)) + continue; + if (first < 0) { + first = cpu; + } else { + second = cpu; + break; + } + } + if (first < 0) { + SCX_ERR("No CPU in affinity mask"); + goto out; + } link = bpf_map__attach_struct_ops(skel->maps.allowed_cpus_ops); - SCX_FAIL_IF(!link, "Failed to attach scheduler"); - - /* Pick an idle CPU from user-space */ - SCX_FAIL_IF(test_select_cpu_from_user(skel), "Failed to pick idle CPU"); - - /* Just sleeping is fine, plenty of scheduling events happening */ + if (!link) { + SCX_ERR("Failed to attach scheduler"); + goto out; + } + + if (test_select_cpu_from_user(skel, "empty mask", -1, true)) + goto out; + + /* A legal candidate may be busy; selection need not succeed. */ + if (test_select_cpu_from_user(skel, "legal candidate", first, false)) + goto out; + + if (second >= 0) { + CPU_ZERO_S(size, pinned); + CPU_SET_S(first, size, pinned); + if (sched_setaffinity(0, size, pinned)) { + SCX_ERR("Failed to pin task (%d)", errno); + goto out; + } + affinity_changed = true; + if (test_select_cpu_from_user(skel, "disjoint masks", second, true)) + goto out; + } else { + fprintf(stderr, "Skipping disjoint masks: need two allowed CPUs\n"); + } + + /* Just sleeping is fine, plenty of scheduling events happening. */ sleep(1); - - SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_NONE)); + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) { + SCX_ERR("Scheduler exited unexpectedly"); + goto out; + } + status = SCX_TEST_PASS; + +out: + if (affinity_changed && sched_setaffinity(0, size, original)) { + SCX_ERR("Failed to restore affinity (%d)", errno); + status = SCX_TEST_FAIL; + } bpf_link__destroy(link); - - return SCX_TEST_PASS; + CPU_FREE(pinned); + CPU_FREE(original); + return status; } static void cleanup(void *ctx) diff --git a/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c new file mode 100644 index 000000000000..841c83abb41b --- /dev/null +++ b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Validate scx_bpf_cgroup_nr_cpus() from both BPF_PROG_TYPE_SYSCALL and + * struct_ops contexts. + * + * Copyright (c) 2026 NVIDIA Corporation. + */ + +#include <scx/common.bpf.h> + +char _license[] SEC("license") = "GPL"; + +UEI_DEFINE(uei); + +/* input to cgroup_nr_cpus_read() */ +u64 query_cgid; +/* output of cgroup_nr_cpus_read(), -1 if @query_cgid couldn't be resolved */ +s64 query_nr_cpus = -1; + +/* recorded by ops.cgroup_init() for @init_cgid */ +u64 init_cgid; +s64 init_nr_cpus = -1; + +SEC("syscall") +int cgroup_nr_cpus_read(void *ctx) +{ + struct cgroup *cgrp; + + query_nr_cpus = -1; + + cgrp = bpf_cgroup_from_id(query_cgid); + if (!cgrp) + return -ENOENT; + + query_nr_cpus = scx_bpf_cgroup_nr_cpus(cgrp); + bpf_cgroup_release(cgrp); + + return 0; +} + +s32 BPF_STRUCT_OPS(cgroup_nr_cpus_cgroup_init, struct cgroup *cgrp, + struct scx_cgroup_init_args *args) +{ + if (cgrp->kn->id == init_cgid) + init_nr_cpus = scx_bpf_cgroup_nr_cpus(cgrp); + + return 0; +} + +void BPF_STRUCT_OPS(cgroup_nr_cpus_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SEC(".struct_ops.link") +struct sched_ext_ops cgroup_nr_cpus_ops = { + .cgroup_init = (void *)cgroup_nr_cpus_cgroup_init, + .exit = (void *)cgroup_nr_cpus_exit, + .name = "cgroup_nr_cpus", +}; diff --git a/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c new file mode 100644 index 000000000000..4973f7080353 --- /dev/null +++ b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c @@ -0,0 +1,419 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Verify that scx_bpf_cgroup_nr_cpus() reports the number of CPUs in a + * cgroup's effective cpuset, including inherited and updated cpusets. + * + * Copyright (c) 2026 NVIDIA Corporation. + */ + +#define _GNU_SOURCE +#include <bpf/bpf.h> +#include <errno.h> +#include <fcntl.h> +#include <limits.h> +#include <linux/limits.h> +#include <scx/common.h> +#include <stdbool.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> + +#include "cgroup_nr_cpus.bpf.skel.h" +#include "cgroup_util.h" +#include "scx_test.h" + +/* + * Hierarchy under the cgroup2 root, all with the cpu controller enabled so + * that ops.cgroup_init() runs for each of them: + * + * parent cpuset enabled by the root, enables cpuset for its children + * parent/child owns a cpuset + * parent/child/leaf no cpuset of its own, inherits child's + */ +struct cgroup_nr_cpus_ctx { + struct cgroup_nr_cpus *skel; + struct bpf_link *link; + char root[PATH_MAX]; + char parent[PATH_MAX]; + char child[PATH_MAX]; + char leaf[PATH_MAX]; + bool parent_created; + bool child_created; + bool leaf_created; +}; + +static int join_path(char *dst, size_t dst_size, const char *parent, const char *name) +{ + int ret; + + ret = snprintf(dst, dst_size, "%s/%s", parent, name); + if (ret < 0 || (size_t)ret >= dst_size) + return -ENAMETOOLONG; + return 0; +} + +static u64 cgroup_id(const char *path) +{ + union { + u64 id; + unsigned char bytes[8]; + } id = {}; + struct file_handle *handle; + int mount_id, ret; + + handle = calloc(1, sizeof(*handle) + sizeof(id)); + if (!handle) + return 0; + handle->handle_bytes = sizeof(id); + ret = name_to_handle_at(AT_FDCWD, path, handle, &mount_id, 0); + if (!ret && handle->handle_bytes == sizeof(id)) + memcpy(id.bytes, handle->f_handle, sizeof(id)); + free(handle); + + return ret ? 0 : id.id; +} + +/* + * Parse a cpulist such as "0-3,8,10-11". Return the number of CPUs and the + * lowest and highest CPU in @first and @last, or -errno on failure. + */ +static int parse_cpulist(const char *cpulist, u32 *first, u32 *last) +{ + const char *p = cpulist; + u32 lowest = UINT_MAX, highest = 0; + int count = 0; + + while (*p && *p != '\n') { + unsigned long start, end_cpu; + char *end; + + errno = 0; + start = strtoul(p, &end, 10); + if (errno || end == p || start > INT_MAX) + return -EINVAL; + end_cpu = start; + p = end; + if (*p == '-') { + end_cpu = strtoul(p + 1, &end, 10); + if (errno || end == p + 1 || end_cpu > INT_MAX || end_cpu < start) + return -EINVAL; + p = end; + } + if (end_cpu - start + 1 > (unsigned long)(INT_MAX - count)) + return -EOVERFLOW; + if (start < lowest) + lowest = start; + if (end_cpu > highest) + highest = end_cpu; + count += end_cpu - start + 1; + if (*p == ',') + p++; + else if (*p && *p != '\n') + return -EINVAL; + } + + if (first) + *first = lowest; + if (last) + *last = highest; + return count; +} + +/* + * Number of CPUs in @cgroup's cpuset.cpus.effective, or -errno. + * + * A sparse cpulist can exceed a page on large systems. cg_read() does a single + * bounded read, so size the buffer for the worst case of NR_CPUS=8192 and + * reject a read that fills it rather than parsing a truncated list. + */ +static int effective_nr_cpus(const char *cgroup, u32 *first, u32 *last) +{ + static char buf[65536]; + + if (cg_read(cgroup, "cpuset.cpus.effective", buf, sizeof(buf))) + return -EIO; + if (strlen(buf) >= sizeof(buf) - 1) + return -EOVERFLOW; + return parse_cpulist(buf, first, last); +} + +/* Run the SYSCALL program to sample scx_bpf_cgroup_nr_cpus() for @path. */ +static int kfunc_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, s64 *nr_cpus) +{ + LIBBPF_OPTS(bpf_test_run_opts, topts); + u64 cgid; + int err; + + cgid = cgroup_id(path); + if (!cgid) { + SCX_ERR("Failed to read cgroup ID of %s", path); + return -ENOENT; + } + + ctx->skel->bss->query_cgid = cgid; + err = bpf_prog_test_run_opts(bpf_program__fd(ctx->skel->progs.cgroup_nr_cpus_read), + &topts); + if (err || topts.retval) { + SCX_ERR("BPF_PROG_RUN failed for %s (err=%d retval=%d)", + path, err, (int)topts.retval); + return err ?: -EIO; + } + + *nr_cpus = ctx->skel->data->query_nr_cpus; + return 0; +} + +static bool check_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, int expected, + const char *what) +{ + s64 nr_cpus; + + if (kfunc_nr_cpus(ctx, path, &nr_cpus)) + return false; + if (nr_cpus != expected) { + SCX_ERR("%s: expected %d CPUs, got %lld", what, expected, + (long long)nr_cpus); + return false; + } + return true; +} + +/* + * Like check_nr_cpus() but tolerate a transient mismatch. A cpuset css being + * disabled stays attached to its cgroup until it's asynchronously offlined, and + * cpuset_num_cpus() keeps reporting its stale mask until then. + */ +static bool wait_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, int expected, + const char *what) +{ + s64 nr_cpus = -1; + int i; + + for (i = 0; i < 1000; i++) { + if (kfunc_nr_cpus(ctx, path, &nr_cpus)) + return false; + if (nr_cpus == expected) + return true; + usleep(1000); + } + SCX_ERR("%s: expected %d CPUs, got %lld", what, expected, (long long)nr_cpus); + return false; +} + +static bool controller_enabled(const char *cgroup, const char *file, const char *controller) +{ + char buf[4096], *saveptr, *token; + + if (cg_read(cgroup, file, buf, sizeof(buf))) + return false; + for (token = strtok_r(buf, "\n ", &saveptr); token; + token = strtok_r(NULL, "\n ", &saveptr)) + if (!strcmp(token, controller)) + return true; + return false; +} + +static void cleanup_ctx(struct cgroup_nr_cpus_ctx *ctx) +{ + bpf_link__destroy(ctx->link); + cgroup_nr_cpus__destroy(ctx->skel); + if (ctx->leaf_created) + cg_destroy(ctx->leaf); + if (ctx->child_created) + cg_destroy(ctx->child); + if (ctx->parent_created) + cg_destroy(ctx->parent); +} + +/* + * Enable @controller in the root's subtree_control if needed. Like the cgroup + * selftests, leave it enabled afterwards: the root is shared, and another + * manager may start relying on the controller while the test runs. + */ +static enum scx_test_status enable_controller(const char *root, const char *controller) +{ + char value[32]; + + if (controller_enabled(root, "cgroup.subtree_control", controller)) + return SCX_TEST_PASS; + if (!controller_enabled(root, "cgroup.controllers", controller)) + return SCX_TEST_SKIP; + + snprintf(value, sizeof(value), "+%s", controller); + if (cg_write(root, "cgroup.subtree_control", value)) + return SCX_TEST_SKIP; + return SCX_TEST_PASS; +} + +static enum scx_test_status setup_cgroups(struct cgroup_nr_cpus_ctx *ctx) +{ + enum scx_test_status status; + char name[64]; + + if (cg_find_unified_root(ctx->root, sizeof(ctx->root), NULL)) + return SCX_TEST_SKIP; + + status = enable_controller(ctx->root, "cpu"); + if (status != SCX_TEST_PASS) + return status; + status = enable_controller(ctx->root, "cpuset"); + if (status != SCX_TEST_PASS) + return status; + + snprintf(name, sizeof(name), "scx_nr_cpus_%d", getpid()); + if (join_path(ctx->parent, sizeof(ctx->parent), ctx->root, name) || + join_path(ctx->child, sizeof(ctx->child), ctx->parent, "child") || + join_path(ctx->leaf, sizeof(ctx->leaf), ctx->child, "leaf")) { + SCX_ERR("Cgroup path is too long"); + return SCX_TEST_FAIL; + } + + if (cg_create(ctx->parent)) { + SCX_ERR("Failed to create cgroup %s", ctx->parent); + return SCX_TEST_FAIL; + } + ctx->parent_created = true; + if (cg_write(ctx->parent, "cgroup.subtree_control", "+cpu +cpuset")) { + SCX_ERR("Failed to enable controllers in %s", ctx->parent); + return SCX_TEST_FAIL; + } + if (cg_create(ctx->child)) { + SCX_ERR("Failed to create cgroup %s", ctx->child); + return SCX_TEST_FAIL; + } + ctx->child_created = true; + if (cg_write(ctx->child, "cgroup.subtree_control", "+cpu")) { + SCX_ERR("Failed to enable cpu in %s", ctx->child); + return SCX_TEST_FAIL; + } + if (cg_create(ctx->leaf)) { + SCX_ERR("Failed to create cgroup %s", ctx->leaf); + return SCX_TEST_FAIL; + } + ctx->leaf_created = true; + + return SCX_TEST_PASS; +} + +static enum scx_test_status run(void *arg) +{ + struct cgroup_nr_cpus_ctx ctx = {}; + enum scx_test_status status; + char value[32]; + u32 first, last; + int nr_root, nr_child; + + (void)arg; + + /* + * SCX_ENUM_INIT() exits the process if vmlinux BTF can't be loaded, so + * run it before creating any cgroups that would then be left behind. + */ + ctx.skel = cgroup_nr_cpus__open(); + if (!ctx.skel) { + SCX_ERR("Failed to open skel"); + return SCX_TEST_FAIL; + } + SCX_ENUM_INIT(ctx.skel); + + status = setup_cgroups(&ctx); + if (status != SCX_TEST_PASS) + goto out; + status = SCX_TEST_FAIL; + + nr_root = effective_nr_cpus(ctx.root, NULL, NULL); + nr_child = effective_nr_cpus(ctx.child, &first, &last); + if (nr_root < 0 || nr_child < 0) { + SCX_ERR("Failed to read effective cpusets"); + goto out; + } + /* The effective cpuset can be empty, e.g. under a partition root. */ + if (nr_child < 2) { + status = SCX_TEST_SKIP; + goto out; + } + + ctx.skel->bss->init_cgid = cgroup_id(ctx.leaf); + if (!ctx.skel->bss->init_cgid) { + SCX_ERR("Failed to read cgroup ID of %s", ctx.leaf); + goto out; + } + if (cgroup_nr_cpus__load(ctx.skel)) { + SCX_ERR("Failed to load skel"); + goto out; + } + + /* The kfunc must be callable without a scheduler attached. */ + if (!check_nr_cpus(&ctx, ctx.root, nr_root, "root") || + !check_nr_cpus(&ctx, ctx.child, nr_child, "child") || + !check_nr_cpus(&ctx, ctx.leaf, nr_child, "inherited leaf")) + goto out; + + /* ops.cgroup_init() runs for existing cgroups when attaching. */ + ctx.link = bpf_map__attach_struct_ops(ctx.skel->maps.cgroup_nr_cpus_ops); + if (!ctx.link) { + SCX_ERR("Failed to attach scheduler"); + goto out; + } + if (ctx.skel->data->init_nr_cpus != nr_child) { + SCX_ERR("ops.cgroup_init(): expected %d CPUs, got %lld", nr_child, + (long long)ctx.skel->data->init_nr_cpus); + goto out; + } + + /* Non-contiguous cpuset, observed by the owner and by the inheritor. */ + if (nr_child > 2) { + snprintf(value, sizeof(value), "%u,%u", first, last); + if (cg_write(ctx.child, "cpuset.cpus", value)) { + SCX_ERR("Failed to set cpuset.cpus=%s for %s", value, ctx.child); + goto out; + } + if (!check_nr_cpus(&ctx, ctx.child, 2, "sparse child") || + !check_nr_cpus(&ctx, ctx.leaf, 2, "sparse inherited leaf")) + goto out; + } + + snprintf(value, sizeof(value), "%u", first); + if (cg_write(ctx.child, "cpuset.cpus", value)) { + SCX_ERR("Failed to set cpuset.cpus=%s for %s", value, ctx.child); + goto out; + } + if (!check_nr_cpus(&ctx, ctx.child, 1, "single-CPU child") || + !check_nr_cpus(&ctx, ctx.leaf, 1, "single-CPU inherited leaf")) + goto out; + + /* + * Disabling cpuset below @parent makes @child and @leaf inherit + * @parent's effective cpuset, which spans all of the root's CPUs. + */ + if (cg_write(ctx.parent, "cgroup.subtree_control", "-cpuset")) { + SCX_ERR("Failed to disable cpuset in %s", ctx.parent); + goto out; + } + nr_child = effective_nr_cpus(ctx.parent, NULL, NULL); + if (nr_child < 0) { + SCX_ERR("Failed to read effective cpuset of %s", ctx.parent); + goto out; + } + if (!wait_nr_cpus(&ctx, ctx.child, nr_child, "child after cpuset disable") || + !wait_nr_cpus(&ctx, ctx.leaf, nr_child, "leaf after cpuset disable")) + goto out; + + if (ctx.skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) { + SCX_ERR("Scheduler exited unexpectedly"); + goto out; + } + + status = SCX_TEST_PASS; +out: + cleanup_ctx(&ctx); + return status; +} + +struct scx_test cgroup_nr_cpus = { + .name = "cgroup_nr_cpus", + .description = "Verify scx_bpf_cgroup_nr_cpus() reports effective cpuset CPU counts", + .run = run, +}; +REGISTER_SCX_TEST(&cgroup_nr_cpus) diff --git a/tools/testing/selftests/sched_ext/config b/tools/testing/selftests/sched_ext/config index aa901b05c8ad..8173f9170ffe 100644 --- a/tools/testing/selftests/sched_ext/config +++ b/tools/testing/selftests/sched_ext/config @@ -1,8 +1,11 @@ CONFIG_SCHED_CLASS_EXT=y CONFIG_CGROUPS=y CONFIG_CGROUP_SCHED=y +CONFIG_CPUSETS=y CONFIG_EXT_GROUP_SCHED=y CONFIG_BPF=y CONFIG_BPF_SYSCALL=y CONFIG_DEBUG_INFO=y CONFIG_DEBUG_INFO_BTF=y +CONFIG_EXPERT=y +CONFIG_SCHED_PROXY_EXEC=y diff --git a/tools/testing/selftests/sched_ext/create_dsq.bpf.c b/tools/testing/selftests/sched_ext/create_dsq.bpf.c index 2cfc4ffd60e2..680cc4b6d8c7 100644 --- a/tools/testing/selftests/sched_ext/create_dsq.bpf.c +++ b/tools/testing/selftests/sched_ext/create_dsq.bpf.c @@ -10,6 +10,8 @@ char _license[] SEC("license") = "GPL"; +u32 nr_lifecycle_tests; + void BPF_STRUCT_OPS(create_dsq_exit_task, struct task_struct *p, struct scx_exit_task_args *args) { @@ -43,7 +45,40 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(create_dsq_init) } bpf_for(i, 0, 1024) { + err = scx_bpf_create_dsq(i, -1); + if (err != -EEXIST) { + scx_bpf_error("Duplicate DSQ %d creation returned %d", i, err); + return -EINVAL; + } + + /* A rejected duplicate must leave the original DSQ accessible. */ + err = scx_bpf_dsq_nr_queued(i); + if (err) { + scx_bpf_error("Original DSQ %d queue count is %d", i, err); + return -EINVAL; + } + + scx_bpf_destroy_dsq(i); + err = scx_bpf_dsq_nr_queued(i); + if (err != -ENOENT) { + scx_bpf_error("Destroyed DSQ %d queue count is %d", i, err); + return -EINVAL; + } + + err = scx_bpf_create_dsq(i, -1); + if (err) { + scx_bpf_error("Failed to recreate DSQ %d: %d", i, err); + return err; + } + + err = scx_bpf_dsq_nr_queued(i); + if (err) { + scx_bpf_error("Recreated DSQ %d queue count is %d", i, err); + return -EINVAL; + } + scx_bpf_destroy_dsq(i); + nr_lifecycle_tests++; } return 0; diff --git a/tools/testing/selftests/sched_ext/create_dsq.c b/tools/testing/selftests/sched_ext/create_dsq.c index d67431f57ac6..422e6532ee71 100644 --- a/tools/testing/selftests/sched_ext/create_dsq.c +++ b/tools/testing/selftests/sched_ext/create_dsq.c @@ -37,6 +37,8 @@ static enum scx_test_status run(void *ctx) bpf_link__destroy(link); + SCX_EQ(skel->bss->nr_lifecycle_tests, 1024); + return SCX_TEST_PASS; } @@ -49,7 +51,7 @@ static void cleanup(void *ctx) struct scx_test create_dsq = { .name = "create_dsq", - .description = "Create and destroy a dsq in a loop", + .description = "Create, reject duplicates, destroy and recreate DSQs", .setup = setup, .run = run, .cleanup = cleanup, diff --git a/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c b/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c new file mode 100644 index 000000000000..983590c6d498 --- /dev/null +++ b/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c @@ -0,0 +1,270 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Verify that ops.dequeue() is called when a task leaves the BPF scheduler's + * custody by being moved to the local DSQ of a CPU other than the one whose + * rq it's on (move_remote_task_to_local_dsq()). + * + * ops.enqueue() puts every task into custody and ops.dispatch() moves it to + * the dispatching CPU's local DSQ, so most moves cross CPUs. With + * @test_use_move_to_local, tasks are queued on a user DSQ and consumed with + * scx_bpf_dsq_move_to_local(). Otherwise, they are queued in a BPF queue and + * dispatched with SCX_DSQ_LOCAL_ON. + * + * Copyright (c) 2026 Google LLC. + */ + +#include <scx/common.bpf.h> + +#define SHARED_DSQ 0 +#define MAX_DISPATCH_POPS 8 + +char _license[] SEC("license") = "GPL"; + +UEI_DEFINE(uei); + +struct { + __uint(type, BPF_MAP_TYPE_QUEUE); + __uint(max_entries, 32768); + __type(value, s32); +} global_queue SEC(".maps"); + +enum task_state { + TASK_NONE = 0, + TASK_ENQUEUED, /* in BPF custody, waiting for ops.dequeue() */ + TASK_DISPATCHED, /* left custody */ +}; + +struct task_ctx { + enum task_state state; + s32 enq_cpu; /* scx_bpf_task_cpu() at ops.enqueue() */ + u64 enqueue_seq; +}; + +struct { + __uint(type, BPF_MAP_TYPE_TASK_STORAGE); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, int); + __type(value, struct task_ctx); +} task_ctx_stor SEC(".maps"); + +/* core_cookie only exists with CONFIG_SCHED_CORE */ +struct task_struct___core_sched { + unsigned long core_cookie; +} __attribute__((preserve_access_index)); + +bool test_use_move_to_local; + +u64 enqueue_cnt, dequeue_cnt, dispatch_dequeue_cnt, change_dequeue_cnt; +u64 remote_dispatch_cnt, remote_running_cnt, missed_dequeue_cnt; +u64 core_sched_exec_dequeue_cnt; + +static struct task_ctx *lookup_task_ctx(struct task_struct *p) +{ + return bpf_task_storage_get(&task_ctx_stor, p, 0, 0); +} + +static bool task_has_core_cookie(struct task_struct *p) +{ + struct task_struct___core_sched *t = (void *)p; + + if (!bpf_core_field_exists(t->core_cookie)) + return false; + return BPF_CORE_READ(t, core_cookie); +} + +s32 BPF_STRUCT_OPS(dequeue_remote_select_cpu, struct task_struct *p, + s32 prev_cpu, u64 wake_flags) +{ + /* no direct dispatch, always go through ops.enqueue() */ + return prev_cpu; +} + +void BPF_STRUCT_OPS(dequeue_remote_enqueue, struct task_struct *p, u64 enq_flags) +{ + struct task_ctx *tctx; + s32 pid = p->pid; + + tctx = lookup_task_ctx(p); + if (!tctx) { + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags); + return; + } + + /* the previous custody period must have ended with ops.dequeue() */ + if (tctx->state == TASK_ENQUEUED) + scx_bpf_error("%d (%s): enqueue while in ENQUEUED state seq=%llu", + p->pid, p->comm, tctx->enqueue_seq); + + /* + * Mark @p as enqueued before making it visible to ops.dispatch() on + * other CPUs, which skips queue entries of tasks not in ENQUEUED + * state as stale. + */ + tctx->state = TASK_ENQUEUED; + tctx->enq_cpu = scx_bpf_task_cpu(p); + tctx->enqueue_seq++; + + if (test_use_move_to_local) { + scx_bpf_dsq_insert(p, SHARED_DSQ, SCX_SLICE_DFL, enq_flags); + } else if (bpf_map_push_elem(&global_queue, &pid, 0)) { + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags); + tctx->state = TASK_DISPATCHED; + tctx->enq_cpu = -1; + goto out; + } + + __sync_fetch_and_add(&enqueue_cnt, 1); +out: + scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_IDLE); +} + +void BPF_STRUCT_OPS(dequeue_remote_dequeue, struct task_struct *p, u64 deq_flags) +{ + struct task_ctx *tctx; + + __sync_fetch_and_add(&dequeue_cnt, 1); + + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + /* + * Only core scheduling can pick a task straight out of custody, and + * only if the task has a core cookie. Otherwise, the custody exit was + * missed when @p was inserted into a local DSQ and got deferred until + * @p was picked. + */ + if ((deq_flags & SCX_DEQ_CORE_SCHED_EXEC) && !task_has_core_cookie(p)) { + __sync_fetch_and_add(&core_sched_exec_dequeue_cnt, 1); + scx_bpf_error("%d (%s): late ops.dequeue() with SCX_DEQ_CORE_SCHED_EXEC (enq_cpu=%d cpu=%d seq=%llu)", + p->pid, p->comm, tctx->enq_cpu, + scx_bpf_task_cpu(p), tctx->enqueue_seq); + } + + /* ops.dequeue() ends the custody period started by ops.enqueue() */ + if (tctx->state != TASK_ENQUEUED) + scx_bpf_error("%d (%s): dequeue outside custody deq_flags=0x%llx state=%d seq=%llu", + p->pid, p->comm, deq_flags, tctx->state, + tctx->enqueue_seq); + + if (deq_flags & SCX_DEQ_SCHED_CHANGE) { + __sync_fetch_and_add(&change_dequeue_cnt, 1); + tctx->state = TASK_NONE; + } else { + __sync_fetch_and_add(&dispatch_dequeue_cnt, 1); + tctx->state = TASK_DISPATCHED; + } +} + +void BPF_STRUCT_OPS(dequeue_remote_dispatch, s32 cpu, struct task_struct *prev) +{ + struct task_ctx *tctx; + struct task_struct *p; + s32 pid; + int i; + + if (test_use_move_to_local) { + scx_bpf_dsq_move_to_local(SHARED_DSQ, 0); + return; + } + + /* pop past stale entries so that they don't leave this CPU idle */ + bpf_for(i, 0, MAX_DISPATCH_POPS) { + if (bpf_map_pop_elem(&global_queue, &pid)) + return; + + p = bpf_task_from_pid(pid); + if (!p) + continue; + + /* + * Entries are stale if @p left custody through a property + * change dequeue or was dispatched from a duplicate entry. + */ + tctx = lookup_task_ctx(p); + if (!tctx || tctx->state != TASK_ENQUEUED) { + bpf_task_release(p); + continue; + } + + if (bpf_cpumask_test_cpu(cpu, p->cpus_ptr)) { + if (scx_bpf_task_cpu(p) != cpu) + __sync_fetch_and_add(&remote_dispatch_cnt, 1); + scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | cpu, + SCX_SLICE_DFL, 0); + } else { + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, 0); + } + + bpf_task_release(p); + return; + } + + /* out of pops with entries left, retry instead of idling this CPU */ + if (!bpf_map_peek_elem(&global_queue, &pid)) + scx_bpf_kick_cpu(cpu, SCX_KICK_IDLE); +} + +void BPF_STRUCT_OPS(dequeue_remote_running, struct task_struct *p) +{ + struct task_ctx *tctx; + + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + /* tasks can only run from a local DSQ, i.e. after leaving custody */ + if (tctx->state == TASK_ENQUEUED) { + __sync_fetch_and_add(&missed_dequeue_cnt, 1); + scx_bpf_error("%d (%s): running without ops.dequeue() (enq_cpu=%d cpu=%d seq=%llu)", + p->pid, p->comm, tctx->enq_cpu, + scx_bpf_task_cpu(p), tctx->enqueue_seq); + return; + } + + if (tctx->enq_cpu >= 0 && tctx->enq_cpu != scx_bpf_task_cpu(p)) + __sync_fetch_and_add(&remote_running_cnt, 1); + tctx->enq_cpu = -1; +} + +s32 BPF_STRUCT_OPS(dequeue_remote_init_task, struct task_struct *p, + struct scx_init_task_args *args) +{ + struct task_ctx *tctx; + + tctx = bpf_task_storage_get(&task_ctx_stor, p, 0, + BPF_LOCAL_STORAGE_GET_F_CREATE); + if (!tctx) + return -ENOMEM; + + /* task storage persists across attachments, start from scratch */ + tctx->state = TASK_NONE; + tctx->enq_cpu = -1; + tctx->enqueue_seq = 0; + + return 0; +} + +s32 BPF_STRUCT_OPS_SLEEPABLE(dequeue_remote_init) +{ + return scx_bpf_create_dsq(SHARED_DSQ, -1); +} + +void BPF_STRUCT_OPS(dequeue_remote_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SEC(".struct_ops.link") +struct sched_ext_ops dequeue_remote_ops = { + .select_cpu = (void *)dequeue_remote_select_cpu, + .enqueue = (void *)dequeue_remote_enqueue, + .dequeue = (void *)dequeue_remote_dequeue, + .dispatch = (void *)dequeue_remote_dispatch, + .running = (void *)dequeue_remote_running, + .init_task = (void *)dequeue_remote_init_task, + .init = (void *)dequeue_remote_init, + .exit = (void *)dequeue_remote_exit, + .flags = SCX_OPS_ENQ_LAST, + .name = "dequeue_remote", +}; diff --git a/tools/testing/selftests/sched_ext/dequeue_remote.c b/tools/testing/selftests/sched_ext/dequeue_remote.c new file mode 100644 index 000000000000..f2c1901030de --- /dev/null +++ b/tools/testing/selftests/sched_ext/dequeue_remote.c @@ -0,0 +1,204 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Verify that ops.dequeue() is called for tasks leaving BPF custody through + * an SCX-internal cross-CPU migration (move_remote_task_to_local_dsq()). + * + * Copyright (c) 2026 Google LLC. + */ +#define _GNU_SOURCE +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <signal.h> +#include <time.h> +#include <sched.h> +#include <bpf/bpf.h> +#include <scx/common.h> +#include <sys/wait.h> +#include "scx_test.h" +#include "dequeue_remote.bpf.skel.h" + +#define MAX_WORKERS 64 +#define RUN_MS 2000 + +static int nr_cpus; + +static long long now_ms(void) +{ + struct timespec ts; + + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000LL + ts.tv_nsec / 1000000; +} + +/* mix of short bursts and sleeps to generate lots of enqueues and wakeups */ +static void worker_fn(int id) +{ + long long end = now_ms() + RUN_MS; + volatile unsigned long sum = 0; + + while (now_ms() < end) { + unsigned long j; + + for (j = 0; j < 20000 + id * 1000; j++) + sum += j; + if (id & 1) + usleep(100); + else + sched_yield(); + } + + exit(0); +} + +static enum scx_test_status run_scenario(struct dequeue_remote *skel, + bool use_move_to_local, + const char *name) +{ + enum scx_test_status ret = SCX_TEST_PASS; + struct bpf_link *link; + pid_t pids[MAX_WORKERS]; + int nr_workers, nr_forked, i; + + nr_workers = 2 * nr_cpus; + if (nr_workers < 4) + nr_workers = 4; + if (nr_workers > MAX_WORKERS) + nr_workers = MAX_WORKERS; + + skel->bss->test_use_move_to_local = use_move_to_local; + skel->bss->enqueue_cnt = 0; + skel->bss->dequeue_cnt = 0; + skel->bss->dispatch_dequeue_cnt = 0; + skel->bss->change_dequeue_cnt = 0; + skel->bss->remote_dispatch_cnt = 0; + skel->bss->remote_running_cnt = 0; + skel->bss->missed_dequeue_cnt = 0; + skel->bss->core_sched_exec_dequeue_cnt = 0; + memset(&skel->data->uei, 0, sizeof(skel->data->uei)); + + link = bpf_map__attach_struct_ops(skel->maps.dequeue_remote_ops); + SCX_FAIL_IF(!link, "Failed to attach struct_ops for %s", name); + + fflush(stdout); + fflush(stderr); + + for (nr_forked = 0; nr_forked < nr_workers; nr_forked++) { + pid_t pid = fork(); + + if (pid < 0) { + SCX_ERR("Failed to fork worker %d", nr_forked); + ret = SCX_TEST_FAIL; + break; + } + if (pid == 0) + worker_fn(nr_forked); + pids[nr_forked] = pid; + } + + /* on failure, kill the remaining workers but still reap them */ + for (i = 0; i < nr_forked; i++) { + int status; + + if (ret != SCX_TEST_PASS) + kill(pids[i], SIGKILL); + + if (waitpid(pids[i], &status, 0) != pids[i]) { + SCX_ERR("Failed to wait for worker %d", i); + ret = SCX_TEST_FAIL; + } else if (ret == SCX_TEST_PASS && status != 0) { + SCX_ERR("Worker %d exited with status %d", i, status); + ret = SCX_TEST_FAIL; + } + } + + bpf_link__destroy(link); + + if (ret != SCX_TEST_PASS) + return ret; + + printf("%s:\n", name); + printf(" workers: %d\n", nr_workers); + printf(" enqueues: %lu\n", (unsigned long)skel->bss->enqueue_cnt); + printf(" dequeues: %lu (dispatch: %lu, property_change: %lu)\n", + (unsigned long)skel->bss->dequeue_cnt, + (unsigned long)skel->bss->dispatch_dequeue_cnt, + (unsigned long)skel->bss->change_dequeue_cnt); + if (!use_move_to_local) + printf(" remote SCX_DSQ_LOCAL_ON dispatches: %lu\n", + (unsigned long)skel->bss->remote_dispatch_cnt); + printf(" ran on a CPU other than the enqueue CPU: %lu\n", + (unsigned long)skel->bss->remote_running_cnt); + printf(" ran without ops.dequeue(): %lu\n", + (unsigned long)skel->bss->missed_dequeue_cnt); + printf(" late SCX_DEQ_CORE_SCHED_EXEC dequeues: %lu\n", + (unsigned long)skel->bss->core_sched_exec_dequeue_cnt); + + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_UNREG)) + SCX_ERR("Scheduler exited with kind=%lld: %s", + (long long)skel->data->uei.kind, skel->data->uei.msg); + SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG)); + + /* the test is meaningless if no task was moved across CPUs */ + SCX_GT(skel->bss->remote_running_cnt, 0); + SCX_EQ(skel->bss->enqueue_cnt, skel->bss->dequeue_cnt); + + return SCX_TEST_PASS; +} + +static enum scx_test_status setup(void **ctx) +{ + struct dequeue_remote *skel; + cpu_set_t cpus; + + SCX_FAIL_IF(sched_getaffinity(0, sizeof(cpus), &cpus), + "Failed to get CPU affinity"); + nr_cpus = CPU_COUNT(&cpus); + if (nr_cpus < 2) { + fprintf(stderr, "Skipping: requires at least 2 usable CPUs\n"); + return SCX_TEST_SKIP; + } + + skel = SCX_OPS_OPEN(dequeue_remote_ops, dequeue_remote); + SCX_OPS_LOAD(skel, dequeue_remote_ops, dequeue_remote, uei); + + *ctx = skel; + + return SCX_TEST_PASS; +} + +static enum scx_test_status run(void *ctx) +{ + struct dequeue_remote *skel = ctx; + enum scx_test_status status; + + status = run_scenario(skel, false, + "BPF queue -> SCX_DSQ_LOCAL_ON | cpu"); + if (status != SCX_TEST_PASS) + return status; + + status = run_scenario(skel, true, + "user DSQ -> scx_bpf_dsq_move_to_local()"); + if (status != SCX_TEST_PASS) + return status; + + return SCX_TEST_PASS; +} + +static void cleanup(void *ctx) +{ + struct dequeue_remote *skel = ctx; + + dequeue_remote__destroy(skel); +} + +struct scx_test dequeue_remote_test = { + .name = "dequeue_remote", + .description = "Verify ops.dequeue() on SCX-internal cross-CPU migrations", + .setup = setup, + .run = run, + .cleanup = cleanup, +}; + +REGISTER_SCX_TEST(&dequeue_remote_test) diff --git a/tools/testing/selftests/sched_ext/enq_blocked.bpf.c b/tools/testing/selftests/sched_ext/enq_blocked.bpf.c new file mode 100644 index 000000000000..212690bf4e07 --- /dev/null +++ b/tools/testing/selftests/sched_ext/enq_blocked.bpf.c @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + * + * Verify that SCX_OPS_ENQ_BLOCKED passes blocked proxy donors through + * ops.enqueue() and record whether callbacks occur on the donor or owner CPU. + */ + +#include <scx/common.bpf.h> + +#define SHARED_DSQ 0 + +char _license[] SEC("license") = "GPL"; + +s32 donor_pid; +s32 donor_cpu = -1; +s32 owner_cpu = -1; +u64 nr_blocked_enqueues; +u64 nr_blocked_enqueues_donor_cpu; +u64 nr_blocked_enqueues_owner_cpu; +u64 nr_blocked_enqueues_other_cpu; +u64 nr_blocked_wakeups; +static u64 vtime_now; + +UEI_DEFINE(uei); + +s32 BPF_STRUCT_OPS(enq_blocked_select_cpu, + struct task_struct *p, s32 prev_cpu, u64 wake_flags) +{ + return prev_cpu; +} + +void BPF_STRUCT_OPS(enq_blocked_enqueue, struct task_struct *p, u64 enq_flags) +{ + u64 vtime = p->scx.dsq_vtime; + + if (enq_flags & SCX_ENQ_BLOCKED) { + int cpu = scx_bpf_task_cpu(p); + + if (enq_flags & SCX_ENQ_WAKEUP) + __sync_fetch_and_add(&nr_blocked_wakeups, 1); + + if (p->pid == donor_pid) { + __sync_fetch_and_add(&nr_blocked_enqueues, 1); + if (cpu == donor_cpu) + __sync_fetch_and_add(&nr_blocked_enqueues_donor_cpu, 1); + else if (cpu == owner_cpu) + __sync_fetch_and_add(&nr_blocked_enqueues_owner_cpu, 1); + else + __sync_fetch_and_add(&nr_blocked_enqueues_other_cpu, 1); + } + } + + /* Limit the amount of budget an idling task can accumulate. */ + if (time_before(vtime, vtime_now - SCX_SLICE_DFL)) + vtime = vtime_now - SCX_SLICE_DFL; + + scx_bpf_dsq_insert_vtime(p, SHARED_DSQ, SCX_SLICE_DFL, vtime, + enq_flags); + scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_IDLE); +} + +void BPF_STRUCT_OPS(enq_blocked_dispatch, s32 cpu, struct task_struct *prev) +{ + scx_bpf_dsq_move_to_local(SHARED_DSQ, 0); +} + +void BPF_STRUCT_OPS(enq_blocked_running, struct task_struct *p) +{ + if (time_before(vtime_now, p->scx.dsq_vtime)) + vtime_now = p->scx.dsq_vtime; +} + +void BPF_STRUCT_OPS(enq_blocked_stopping, struct task_struct *p, bool runnable) +{ + u64 delta = scale_by_task_weight_inverse(p, + SCX_SLICE_DFL - p->scx.slice); + + scx_bpf_task_set_dsq_vtime(p, p->scx.dsq_vtime + delta); +} + +void BPF_STRUCT_OPS(enq_blocked_enable, struct task_struct *p) +{ + scx_bpf_task_set_dsq_vtime(p, vtime_now); +} + +s32 BPF_STRUCT_OPS_SLEEPABLE(enq_blocked_init) +{ + int ret; + + ret = scx_bpf_create_dsq(SHARED_DSQ, -1); + if (ret) { + scx_bpf_error("failed to create DSQ %d (%d)", SHARED_DSQ, ret); + return ret; + } + + return 0; +} + +void BPF_STRUCT_OPS(enq_blocked_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SEC(".struct_ops.link") +struct sched_ext_ops enq_blocked_ops = { + .select_cpu = (void *)enq_blocked_select_cpu, + .enqueue = (void *)enq_blocked_enqueue, + .dispatch = (void *)enq_blocked_dispatch, + .running = (void *)enq_blocked_running, + .stopping = (void *)enq_blocked_stopping, + .enable = (void *)enq_blocked_enable, + .init = (void *)enq_blocked_init, + .exit = (void *)enq_blocked_exit, + .name = "enq_blocked", +}; diff --git a/tools/testing/selftests/sched_ext/enq_blocked.c b/tools/testing/selftests/sched_ext/enq_blocked.c new file mode 100644 index 000000000000..26204548bbf0 --- /dev/null +++ b/tools/testing/selftests/sched_ext/enq_blocked.c @@ -0,0 +1,917 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + * + * Exercise a priority inversion with the owner and donor first pinned to the + * same CPU, then with each on a different CPU. A high-priority donor blocks on + * a mutex held by a low-priority owner while one medium-priority contender per + * available CPU keeps the system busy. A weighted-vruntime BPF scheduler runs + * both CPU placement configurations with SCX_OPS_ENQ_BLOCKED first disabled + * and then enabled. The test validates blocked-donor admission and reports the + * average mutex hold and wait times, plus their enabled-minus-disabled deltas, + * for each configuration. The timing data is informational. + * + * CONFIG_SCHED_PROXY_EXEC=y is required to exercise the proxy-execution paths. + */ +#define _GNU_SOURCE + +#include <bpf/bpf.h> +#include <errno.h> +#include <fcntl.h> +#include <limits.h> +#include <pthread.h> +#include <sched.h> +#include <scx/common.h> +#include <stdatomic.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/ioctl.h> +#include <sys/resource.h> +#include <sys/syscall.h> +#include <time.h> +#include <unistd.h> + +#include "enq_blocked.bpf.skel.h" +#include "enq_blocked.h" +#include "scx_test.h" + +#define MODULE_NAME "scx_enq_blocked_test" +#define MODULE_FILE "test_modules/" MODULE_NAME ".ko" +#define DEVICE_PATH "/dev/scx_enq_blocked" +#define WAIT_STEP_US 1000 +#define WAIT_TIMEOUT_MS 2000 +#define NR_WARMUP_TRIALS 1 +#define NR_MEASURED_TRIALS 10 +#define NR_TRIALS (NR_WARMUP_TRIALS + NR_MEASURED_TRIALS) +#define JOIN_TIMEOUT_MS ((NR_TRIALS + 1) * WAIT_TIMEOUT_MS) +#define OWNER_NICE 19 +#define DONOR_NICE -20 +#define CONTENDER_NICE 0 + +struct thread_ctx { + atomic_bool start_donor; + atomic_bool abort; + atomic_bool stop_contender; + atomic_bool measurement_ready; + atomic_int donor_pid; + atomic_int donor_completed; + int fd; + int donor_cpu; + int owner_cpu; +}; + +struct contender_ctx { + struct thread_ctx *thread_ctx; + atomic_int status; + int cpu; +}; + +struct run_result { + struct enq_blocked_stats stats; + u64 nr_blocked_enqueues; + u64 nr_blocked_enqueues_donor_cpu; + u64 nr_blocked_enqueues_owner_cpu; + u64 nr_blocked_enqueues_other_cpu; + u64 nr_blocked_wakeups; +}; + +static bool parse_bool(const char *value, bool *result) +{ + if (!strcasecmp(value, "1") || !strcasecmp(value, "y") || + !strcasecmp(value, "yes") || !strcasecmp(value, "on") || + !strcasecmp(value, "true")) { + *result = true; + return true; + } + + if (!strcasecmp(value, "0") || !strcasecmp(value, "n") || + !strcasecmp(value, "no") || !strcasecmp(value, "off") || + !strcasecmp(value, "false")) { + *result = false; + return true; + } + + return false; +} + +static bool cmdline_bool(const char *name, bool default_value) +{ + char cmdline[4096], *newline, *saveptr = NULL, *token; + size_t name_len = strlen(name); + bool value = default_value; + FILE *file; + + file = fopen("/proc/cmdline", "r"); + if (!file) + return default_value; + + if (!fgets(cmdline, sizeof(cmdline), file)) { + fclose(file); + return default_value; + } + fclose(file); + newline = strchr(cmdline, '\n'); + if (newline) + *newline = '\0'; + + for (token = strtok_r(cmdline, " ", &saveptr); token; + token = strtok_r(NULL, " ", &saveptr)) { + bool parsed; + + if (strncmp(token, name, name_len) || token[name_len] != '=') + continue; + if (parse_bool(token + name_len + 1, &parsed)) + value = parsed; + } + + return value; +} + +static int module_path(char *path, size_t size) +{ + ssize_t len; + char *slash; + + len = readlink("/proc/self/exe", path, size - 1); + if (len < 0) + return -errno; + path[len] = '\0'; + + slash = strrchr(path, '/'); + if (!slash) + return -EINVAL; + *slash = '\0'; + + if (snprintf(slash, size - (slash - path), "/%s", MODULE_FILE) >= + size - (slash - path)) + return -ENAMETOOLONG; + + return 0; +} + +static int load_test_module(bool *loaded_here) +{ + char path[PATH_MAX]; + int fd, err; + + err = module_path(path, sizeof(path)); + if (err) + return err; + + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) + return -errno; + + if (syscall(SYS_finit_module, fd, "", 0)) { + err = errno; + close(fd); + if (err == EEXIST) + return 0; + return -err; + } + + close(fd); + *loaded_here = true; + return 0; +} + +static void unload_test_module(bool loaded_here) +{ + if (loaded_here && syscall(SYS_delete_module, MODULE_NAME, O_NONBLOCK)) + SCX_ERR("Failed to unload %s (%d)", MODULE_NAME, errno); +} + +static int pin_to_cpu(int cpu) +{ + cpu_set_t mask; + + CPU_ZERO(&mask); + CPU_SET(cpu, &mask); + return sched_setaffinity(0, sizeof(mask), &mask) ? errno : 0; +} + +static int select_test_cpus(bool cross_cpu, cpu_set_t *mask, int *donor_cpu, + int *owner_cpu) +{ + int cpu, first = -1; + + if (sched_getaffinity(0, sizeof(*mask), mask)) + return -errno; + + for (cpu = 0; cpu < CPU_SETSIZE; cpu++) { + if (!CPU_ISSET(cpu, mask)) + continue; + if (first < 0) { + first = cpu; + if (!cross_cpu) + break; + } else { + *donor_cpu = first; + *owner_cpu = cpu; + return 0; + } + } + + if (first < 0) + return -ENODEV; + if (cross_cpu) + return -EAGAIN; + + *donor_cpu = first; + *owner_cpu = first; + return 0; +} + +static int set_nice(int nice) +{ + return setpriority(PRIO_PROCESS, 0, nice) ? errno : 0; +} + +static bool wait_for_pid(atomic_int *pid) +{ + int waited_ms; + + for (waited_ms = 0; waited_ms < WAIT_TIMEOUT_MS; waited_ms++) { + if (atomic_load_explicit(pid, memory_order_acquire) > 0) + return true; + usleep(WAIT_STEP_US); + } + + return false; +} + +static int wait_for_contenders(struct contender_ctx *contenders, + size_t nr_contenders) +{ + size_t i, nr_ready; + int status, waited_ms; + + for (waited_ms = 0; waited_ms < WAIT_TIMEOUT_MS; waited_ms++) { + nr_ready = 0; + for (i = 0; i < nr_contenders; i++) { + status = atomic_load_explicit(&contenders[i].status, + memory_order_acquire); + if (status < 0) + return status; + if (status > 0) + nr_ready++; + } + if (nr_ready == nr_contenders) + return 1; + usleep(WAIT_STEP_US); + } + + return -ETIMEDOUT; +} + +static int wait_for_donor_state(struct thread_ctx *ctx, int expected) +{ + int state, waited_ms; + + for (waited_ms = 0; waited_ms < WAIT_TIMEOUT_MS; waited_ms++) { + state = ioctl(ctx->fd, ENQ_BLOCKED_IOCTL_DONOR_STATE); + if (state == expected) + return state; + if (state < 0 && errno != ENOENT) + return -errno; + usleep(WAIT_STEP_US); + } + + return -ETIMEDOUT; +} + +static bool wait_for_donor(struct thread_ctx *ctx, int trial) +{ + int waited_ms; + + for (waited_ms = 0; waited_ms < WAIT_TIMEOUT_MS; waited_ms++) { + if (atomic_load_explicit(&ctx->donor_completed, + memory_order_acquire) >= trial) + return true; + if (atomic_load_explicit(&ctx->abort, memory_order_relaxed)) + return false; + usleep(WAIT_STEP_US); + } + + return false; +} + +static bool wait_for_measurement(struct thread_ctx *ctx) +{ + while (!atomic_load_explicit(&ctx->measurement_ready, + memory_order_acquire) && + !atomic_load_explicit(&ctx->abort, memory_order_relaxed)) + sched_yield(); + + return !atomic_load_explicit(&ctx->abort, memory_order_relaxed); +} + +static void *contender_fn(void *arg) +{ + struct contender_ctx *contender = arg; + struct thread_ctx *ctx = contender->thread_ctx; + int err; + + err = pin_to_cpu(contender->cpu); + if (!err) + err = set_nice(CONTENDER_NICE); + atomic_store_explicit(&contender->status, err ? -err : 1, + memory_order_release); + if (err) + return (void *)(uintptr_t)err; + + while (!atomic_load_explicit(&ctx->stop_contender, + memory_order_relaxed)) + ; + + return NULL; +} + +static void *owner_fn(void *arg) +{ + struct thread_ctx *ctx = arg; + int err, i; + + err = pin_to_cpu(ctx->owner_cpu); + if (err) + return (void *)(uintptr_t)err; + err = set_nice(OWNER_NICE); + if (err) + return (void *)(uintptr_t)err; + + for (i = 0; i < NR_TRIALS; i++) { + if (ioctl(ctx->fd, ENQ_BLOCKED_IOCTL_OWNER)) + return (void *)(uintptr_t)errno; + if (!wait_for_donor(ctx, i + 1)) + return (void *)(uintptr_t)ETIMEDOUT; + + if (i + 1 == NR_WARMUP_TRIALS && !wait_for_measurement(ctx)) + return NULL; + } + + return NULL; +} + +static int run_donor_trial(struct thread_ctx *ctx) +{ + int waited_ms; + + for (waited_ms = 0; waited_ms < WAIT_TIMEOUT_MS; waited_ms++) { + if (!ioctl(ctx->fd, ENQ_BLOCKED_IOCTL_DONOR)) + return 0; + if (errno != EAGAIN) + return -errno; + usleep(WAIT_STEP_US); + } + + return -ETIMEDOUT; +} + +static void *donor_fn(void *arg) +{ + struct thread_ctx *ctx = arg; + int err, i; + + err = pin_to_cpu(ctx->donor_cpu); + if (err) + return (void *)(uintptr_t)err; + err = set_nice(DONOR_NICE); + if (err) + return (void *)(uintptr_t)err; + + atomic_store_explicit(&ctx->donor_pid, syscall(SYS_gettid), + memory_order_release); + while (!atomic_load_explicit(&ctx->start_donor, memory_order_acquire) && + !atomic_load_explicit(&ctx->abort, memory_order_relaxed)) + sched_yield(); + + if (atomic_load_explicit(&ctx->abort, memory_order_relaxed)) + return NULL; + + for (i = 0; i < NR_TRIALS; i++) { + err = run_donor_trial(ctx); + if (err) + return (void *)(uintptr_t)-err; + atomic_store_explicit(&ctx->donor_completed, i + 1, + memory_order_release); + } + + return NULL; +} + +static void print_avg_time(const char *name, u64 total_ns, u64 samples) +{ + u64 avg_ns = samples ? total_ns / samples : 0; + + printf(" %s_avg_ns=%llu (%llu.%03llu ms, samples=%llu)\n", name, + (unsigned long long)avg_ns, + (unsigned long long)(avg_ns / 1000000), + (unsigned long long)((avg_ns / 1000) % 1000), + (unsigned long long)samples); +} + +static void print_avg_delta(const char *name, u64 disabled_total, + u64 disabled_samples, u64 enabled_total, + u64 enabled_samples) +{ + u64 disabled_avg, enabled_avg; + s64 delta_ns; + double delta_pct; + + if (!disabled_samples || !enabled_samples) + return; + + disabled_avg = disabled_total / disabled_samples; + enabled_avg = enabled_total / enabled_samples; + delta_ns = (s64)enabled_avg - (s64)disabled_avg; + delta_pct = disabled_avg ? 100.0 * delta_ns / disabled_avg : 0.0; + + printf(" %s_delta_ns=%+lld (%+.2f%%)\n", name, + (long long)delta_ns, delta_pct); +} + +static int join_thread(pthread_t thread, const struct timespec *deadline, + int *thread_err) +{ + void *result; + int err; + + err = pthread_timedjoin_np(thread, &result, deadline); + if (err) + return err; + + *thread_err = (int)(uintptr_t)result; + return 0; +} + +static void set_join_deadline(struct timespec *deadline) +{ + clock_gettime(CLOCK_REALTIME, deadline); + deadline->tv_sec += JOIN_TIMEOUT_MS / 1000; + deadline->tv_nsec += (JOIN_TIMEOUT_MS % 1000) * 1000000; + if (deadline->tv_nsec >= 1000000000) { + deadline->tv_sec++; + deadline->tv_nsec -= 1000000000; + } +} + +static enum scx_test_status setup(void **ctx) +{ + struct enq_blocked *skel; + u64 flag; + + skel = enq_blocked__open(); + SCX_FAIL_IF(!skel, "Failed to open skel"); + SCX_ENUM_INIT(skel); + + flag = SCX_OPS_ENQ_BLOCKED; + if (!flag) { + enq_blocked__destroy(skel); + fprintf(stderr, "SKIP: SCX_OPS_ENQ_BLOCKED is unavailable\n"); + return SCX_TEST_SKIP; + } + + enq_blocked__destroy(skel); + *ctx = NULL; + return SCX_TEST_PASS; +} + +static enum scx_test_status run_one(bool enq_blocked, bool cross_cpu, + struct run_result *result) +{ + struct enq_blocked *skel; + struct thread_ctx thread_ctx = {}; + struct contender_ctx *contender_ctxs = NULL; + struct bpf_link *link = NULL; + pthread_t owner, donor, *contenders = NULL; + struct timespec join_deadline; + cpu_set_t allowed_cpus; + bool module_loaded = false; + bool owner_started = false, donor_started = false; + bool join_timed_out = false; + bool proxy_enabled; + enum scx_test_status status = SCX_TEST_PASS; + int cpu, donor_pid, donor_state, err, thread_err; + size_t i, nr_contenders, nr_contenders_started = 0; + size_t nr_contenders_joined = 0; + u64 nr_blocked, nr_blocked_donor_cpu, nr_blocked_owner_cpu; + u64 nr_blocked_other_cpu, nr_blocked_wakeups; + struct enq_blocked_stats stats; + + err = select_test_cpus(cross_cpu, &allowed_cpus, &thread_ctx.donor_cpu, + &thread_ctx.owner_cpu); + if (err == -EAGAIN) { + fprintf(stderr, "SKIP: cross-CPU case requires two allowed CPUs\n"); + return SCX_TEST_SKIP; + } + if (err) { + SCX_ERR("Failed to select test CPUs (%d)", -err); + return SCX_TEST_FAIL; + } + nr_contenders = CPU_COUNT(&allowed_cpus); + contenders = calloc(nr_contenders, sizeof(*contenders)); + contender_ctxs = calloc(nr_contenders, sizeof(*contender_ctxs)); + if (!contenders || !contender_ctxs) { + SCX_ERR("Failed to allocate %zu contender threads", nr_contenders); + status = SCX_TEST_FAIL; + goto out_contenders; + } + + skel = enq_blocked__open(); + if (!skel) { + SCX_ERR("Failed to open skel"); + status = SCX_TEST_FAIL; + goto out_contenders; + } + SCX_ENUM_INIT(skel); + skel->struct_ops.enq_blocked_ops->flags = + SCX_OPS_ENQ_LAST | + (enq_blocked ? SCX_OPS_ENQ_BLOCKED : 0); + if (enq_blocked__load(skel)) { + SCX_ERR("Failed to load skel"); + status = SCX_TEST_FAIL; + goto out_skel; + } + + err = load_test_module(&module_loaded); + if (err == -EPERM || err == -ENOENT) { + fprintf(stderr, "SKIP: cannot load mutex fixture (%d)\n", -err); + status = SCX_TEST_SKIP; + goto out_skel; + } + if (err) { + SCX_ERR("Failed to load mutex fixture (%d)", -err); + status = SCX_TEST_FAIL; + goto out_skel; + } + + thread_ctx.fd = open(DEVICE_PATH, O_RDONLY | O_CLOEXEC); + if (thread_ctx.fd < 0) { + SCX_ERR("Failed to open %s (%d)", DEVICE_PATH, errno); + status = SCX_TEST_FAIL; + goto out_module; + } + err = ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_PROXY_SUPPORTED); + if (err < 0) { + SCX_ERR("Failed to query proxy-exec support (%d)", errno); + status = SCX_TEST_FAIL; + goto out_fd; + } + proxy_enabled = err && cmdline_bool("sched_proxy_exec", true); + if (!proxy_enabled) { + fprintf(stderr, "SKIP: proxy execution is not enabled\n"); + status = SCX_TEST_SKIP; + goto out_fd; + } + if (ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_RESET_STATS)) { + SCX_ERR("Failed to reset mutex statistics (%d)", errno); + status = SCX_TEST_FAIL; + goto out_fd; + } + + if (ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_PREP_ATTACH)) { + SCX_ERR("Failed to prepare scheduler attachment (%d)", errno); + status = SCX_TEST_FAIL; + goto out_fd; + } + + err = pthread_create(&owner, NULL, owner_fn, &thread_ctx); + if (err) { + SCX_ERR("Failed to create owner thread (%d)", err); + status = SCX_TEST_FAIL; + goto out; + } + owner_started = true; + + err = pthread_create(&donor, NULL, donor_fn, &thread_ctx); + if (err) { + SCX_ERR("Failed to create donor thread (%d)", err); + status = SCX_TEST_FAIL; + goto out; + } + donor_started = true; + + if (!wait_for_pid(&thread_ctx.donor_pid)) { + SCX_ERR("Timed out waiting for donor thread"); + status = SCX_TEST_FAIL; + goto out; + } + + donor_pid = atomic_load_explicit(&thread_ctx.donor_pid, + memory_order_acquire); + skel->bss->donor_pid = donor_pid; + skel->data->donor_cpu = thread_ctx.donor_cpu; + skel->data->owner_cpu = thread_ctx.owner_cpu; + atomic_store_explicit(&thread_ctx.start_donor, true, + memory_order_release); + + donor_state = ENQ_BLOCKED_DONOR_SLEEPING; + if (proxy_enabled) + donor_state |= ENQ_BLOCKED_DONOR_ON_RQ; + err = wait_for_donor_state(&thread_ctx, donor_state); + if (err < 0) { + SCX_ERR("Donor did not block before scheduler attachment (%d)", -err); + status = SCX_TEST_FAIL; + goto out; + } + + link = bpf_map__attach_struct_ops(skel->maps.enq_blocked_ops); + if (!link) { + SCX_ERR("Failed to attach scheduler"); + status = SCX_TEST_FAIL; + goto out; + } + + /* Scheduler ownership changes start from a fully blocked donor. */ + donor_state = ENQ_BLOCKED_DONOR_SLEEPING; + err = wait_for_donor_state(&thread_ctx, donor_state); + if (err < 0) { + SCX_ERR("Unexpected donor state after scheduler attachment (%d)", + -err); + status = SCX_TEST_FAIL; + goto out; + } + + if (ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_ATTACH_DONE)) { + SCX_ERR("Failed to complete scheduler attachment (%d)", errno); + status = SCX_TEST_FAIL; + goto out; + } + + i = 0; + for (cpu = 0; cpu < CPU_SETSIZE; cpu++) { + if (!CPU_ISSET(cpu, &allowed_cpus)) + continue; + + contender_ctxs[i].thread_ctx = &thread_ctx; + contender_ctxs[i].cpu = cpu; + atomic_init(&contender_ctxs[i].status, 0); + err = pthread_create(&contenders[i], NULL, contender_fn, + &contender_ctxs[i]); + if (err) { + SCX_ERR("Failed to create contender for CPU %d (%d)", + cpu, err); + status = SCX_TEST_FAIL; + goto out; + } + nr_contenders_started++; + i++; + } + + err = wait_for_contenders(contender_ctxs, nr_contenders); + if (err != 1) { + SCX_ERR("Contender threads failed (%d)", -err); + status = SCX_TEST_FAIL; + goto out; + } + + /* + * The first trial spans scheduler attachment and validates the state + * transition, but including it would skew scheduling latency. Exclude it + * from both the mutex and BPF enqueue measurements. + */ + if (!wait_for_donor(&thread_ctx, NR_WARMUP_TRIALS)) { + SCX_ERR("Timed out waiting for warm-up trial"); + status = SCX_TEST_FAIL; + goto out; + } + if (ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_RESET_STATS)) { + SCX_ERR("Failed to reset mutex statistics after warm-up (%d)", + errno); + status = SCX_TEST_FAIL; + goto out; + } + skel->bss->nr_blocked_enqueues = 0; + skel->bss->nr_blocked_enqueues_donor_cpu = 0; + skel->bss->nr_blocked_enqueues_owner_cpu = 0; + skel->bss->nr_blocked_enqueues_other_cpu = 0; + skel->bss->nr_blocked_wakeups = 0; + atomic_store_explicit(&thread_ctx.measurement_ready, true, + memory_order_release); + +out: + ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_ATTACH_DONE); + if (status != SCX_TEST_PASS) { + atomic_store_explicit(&thread_ctx.abort, true, memory_order_release); + atomic_store_explicit(&thread_ctx.start_donor, true, + memory_order_release); + atomic_store_explicit(&thread_ctx.measurement_ready, true, + memory_order_release); + } + + set_join_deadline(&join_deadline); + if (donor_started) { + err = join_thread(donor, &join_deadline, &thread_err); + if (err == ETIMEDOUT) { + SCX_ERR("Timed out waiting for donor thread"); + join_timed_out = true; + status = SCX_TEST_FAIL; + } else if (err) { + SCX_ERR("Failed to join donor thread (%d)", err); + status = SCX_TEST_FAIL; + } else { + donor_started = false; + if (thread_err) { + SCX_ERR("Donor thread failed (%d)", thread_err); + status = SCX_TEST_FAIL; + } + } + } + if (!join_timed_out && owner_started) { + err = join_thread(owner, &join_deadline, &thread_err); + if (err == ETIMEDOUT) { + SCX_ERR("Timed out waiting for owner thread"); + join_timed_out = true; + status = SCX_TEST_FAIL; + } else if (err) { + SCX_ERR("Failed to join owner thread (%d)", err); + status = SCX_TEST_FAIL; + } else { + owner_started = false; + if (thread_err) { + SCX_ERR("Owner thread failed (%d)", thread_err); + status = SCX_TEST_FAIL; + } + } + } + atomic_store_explicit(&thread_ctx.stop_contender, true, + memory_order_release); + for (i = 0; !join_timed_out && i < nr_contenders_started; i++) { + err = join_thread(contenders[i], &join_deadline, &thread_err); + if (err == ETIMEDOUT) { + SCX_ERR("Timed out waiting for contender on CPU %d", + contender_ctxs[i].cpu); + join_timed_out = true; + status = SCX_TEST_FAIL; + } else if (err) { + SCX_ERR("Failed to join contender on CPU %d (%d)", + contender_ctxs[i].cpu, err); + status = SCX_TEST_FAIL; + } else { + nr_contenders_joined++; + if (thread_err) { + SCX_ERR("Contender on CPU %d failed (%d)", + contender_ctxs[i].cpu, thread_err); + status = SCX_TEST_FAIL; + } + } + } + + /* Restore the fair scheduler before waiting for any stranded thread. */ + if (join_timed_out) { + atomic_store_explicit(&thread_ctx.abort, true, + memory_order_release); + if (link) { + bpf_link__destroy(link); + link = NULL; + } + if (donor_started) + pthread_join(donor, NULL); + if (owner_started) + pthread_join(owner, NULL); + for (i = nr_contenders_joined; + i < nr_contenders_started; i++) + pthread_join(contenders[i], NULL); + } + + if (ioctl(thread_ctx.fd, ENQ_BLOCKED_IOCTL_GET_STATS, &stats)) { + SCX_ERR("Failed to read mutex statistics (%d)", errno); + status = SCX_TEST_FAIL; + } else { + result->stats = stats; + printf("\n[topology=%s SCX_OPS_ENQ_BLOCKED=%s]\n", + cross_cpu ? "cross-cpu" : "same-cpu", + enq_blocked ? "enabled" : "disabled"); + printf(" proxy_exec=%s\n", + proxy_enabled ? "enabled" : "disabled"); + printf(" donor_cpu=%d\n", thread_ctx.donor_cpu); + printf(" owner_cpu=%d\n", thread_ctx.owner_cpu); + printf(" nr_contenders=%zu\n", nr_contenders); + printf(" measured_trials=%d\n", NR_MEASURED_TRIALS); + printf(" owner_nice=%d\n", OWNER_NICE); + printf(" donor_nice=%d\n", DONOR_NICE); + printf(" contender_nice=%d\n", CONTENDER_NICE); + print_avg_time("mutex_hold", stats.hold_time_ns, stats.nr_holds); + print_avg_time("mutex_wait", stats.wait_time_ns, stats.nr_waits); + if (stats.nr_holds != NR_MEASURED_TRIALS || + stats.nr_waits != NR_MEASURED_TRIALS) { + SCX_ERR("Expected %d measured trials, got %llu holds and %llu waits", + NR_MEASURED_TRIALS, + (unsigned long long)stats.nr_holds, + (unsigned long long)stats.nr_waits); + status = SCX_TEST_FAIL; + } + } + + nr_blocked = skel->bss->nr_blocked_enqueues; + nr_blocked_donor_cpu = skel->bss->nr_blocked_enqueues_donor_cpu; + nr_blocked_owner_cpu = skel->bss->nr_blocked_enqueues_owner_cpu; + nr_blocked_other_cpu = skel->bss->nr_blocked_enqueues_other_cpu; + nr_blocked_wakeups = skel->bss->nr_blocked_wakeups; + result->nr_blocked_enqueues = nr_blocked; + result->nr_blocked_enqueues_donor_cpu = nr_blocked_donor_cpu; + result->nr_blocked_enqueues_owner_cpu = nr_blocked_owner_cpu; + result->nr_blocked_enqueues_other_cpu = nr_blocked_other_cpu; + result->nr_blocked_wakeups = nr_blocked_wakeups; + printf(" nr_blocked_enqueues=%llu\n", + (unsigned long long)nr_blocked); + printf(" nr_blocked_enqueues_donor_cpu=%llu\n", + (unsigned long long)nr_blocked_donor_cpu); + printf(" nr_blocked_enqueues_owner_cpu=%llu\n", + (unsigned long long)nr_blocked_owner_cpu); + printf(" nr_blocked_enqueues_other_cpu=%llu\n", + (unsigned long long)nr_blocked_other_cpu); + printf(" nr_blocked_wakeups=%llu\n", + (unsigned long long)nr_blocked_wakeups); + if (status == SCX_TEST_PASS) { + if (enq_blocked && proxy_enabled && !nr_blocked) { + SCX_ERR("ops.enqueue() did not receive the blocked donor"); + status = SCX_TEST_FAIL; + } else if ((!enq_blocked || !proxy_enabled) && nr_blocked) { + SCX_ERR("ops.enqueue() unexpectedly received %llu blocked donors", + (unsigned long long)nr_blocked); + status = SCX_TEST_FAIL; + } else if (nr_blocked_wakeups) { + SCX_ERR("Ordinary wakeups received %llu blocked enqueue flags", + (unsigned long long)nr_blocked_wakeups); + status = SCX_TEST_FAIL; + } else if (nr_blocked_other_cpu) { + SCX_ERR("Blocked donor had %llu enqueues on unexpected CPUs", + (unsigned long long)nr_blocked_other_cpu); + status = SCX_TEST_FAIL; + } + } + + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) { + SCX_ERR("Scheduler exited unexpectedly (kind=%llu code=%lld)", + (unsigned long long)skel->data->uei.kind, + (long long)skel->data->uei.exit_code); + status = SCX_TEST_FAIL; + } + + if (link) + bpf_link__destroy(link); +out_fd: + close(thread_ctx.fd); +out_module: + unload_test_module(module_loaded); +out_skel: + enq_blocked__destroy(skel); +out_contenders: + free(contender_ctxs); + free(contenders); + return status; +} + +static enum scx_test_status run_topology(bool cross_cpu) +{ + struct run_result disabled = {}, enabled = {}; + enum scx_test_status status; + + status = run_one(false, cross_cpu, &disabled); + if (status != SCX_TEST_PASS) + return status; + + status = run_one(true, cross_cpu, &enabled); + if (status != SCX_TEST_PASS) + return status; + + printf("\n[topology=%s delta: enabled - disabled]\n", + cross_cpu ? "cross-cpu" : "same-cpu"); + print_avg_delta("mutex_hold", disabled.stats.hold_time_ns, + disabled.stats.nr_holds, enabled.stats.hold_time_ns, + enabled.stats.nr_holds); + print_avg_delta("mutex_wait", disabled.stats.wait_time_ns, + disabled.stats.nr_waits, enabled.stats.wait_time_ns, + enabled.stats.nr_waits); + + return SCX_TEST_PASS; +} + +static enum scx_test_status run(void *ctx) +{ + enum scx_test_status status; + + (void)ctx; + + status = run_topology(false); + if (status != SCX_TEST_PASS) + return status; + + status = run_topology(true); + if (status == SCX_TEST_SKIP) + return SCX_TEST_PASS; + + return status; +} + +struct scx_test enq_blocked = { + .name = "enq_blocked", + .description = "Verify proxy donor admission under CPU-wide contention", + .setup = setup, + .run = run, +}; + +REGISTER_SCX_TEST(&enq_blocked) diff --git a/tools/testing/selftests/sched_ext/enq_blocked.h b/tools/testing/selftests/sched_ext/enq_blocked.h new file mode 100644 index 000000000000..ef1eb97feebe --- /dev/null +++ b/tools/testing/selftests/sched_ext/enq_blocked.h @@ -0,0 +1,28 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES */ +#ifndef __ENQ_BLOCKED_H +#define __ENQ_BLOCKED_H + +#include <linux/ioctl.h> +#include <linux/types.h> + +struct enq_blocked_stats { + __u64 hold_time_ns; + __u64 wait_time_ns; + __u64 nr_holds; + __u64 nr_waits; +}; + +#define ENQ_BLOCKED_IOCTL_OWNER _IO('s', 1) +#define ENQ_BLOCKED_IOCTL_DONOR _IO('s', 2) +#define ENQ_BLOCKED_IOCTL_RESET_STATS _IO('s', 3) +#define ENQ_BLOCKED_IOCTL_GET_STATS _IOR('s', 4, struct enq_blocked_stats) +#define ENQ_BLOCKED_IOCTL_PREP_ATTACH _IO('s', 5) +#define ENQ_BLOCKED_IOCTL_ATTACH_DONE _IO('s', 6) +#define ENQ_BLOCKED_IOCTL_DONOR_STATE _IO('s', 7) +#define ENQ_BLOCKED_IOCTL_PROXY_SUPPORTED _IO('s', 8) + +#define ENQ_BLOCKED_DONOR_SLEEPING (1U << 0) +#define ENQ_BLOCKED_DONOR_ON_RQ (1U << 1) + +#endif /* __ENQ_BLOCKED_H */ diff --git a/tools/testing/selftests/sched_ext/hotplug.c b/tools/testing/selftests/sched_ext/hotplug.c index 0cfbb111a2d0..10b8d42bd89b 100644 --- a/tools/testing/selftests/sched_ext/hotplug.c +++ b/tools/testing/selftests/sched_ext/hotplug.c @@ -21,15 +21,17 @@ static bool is_cpu_online(void) return file_read_long(online_path) > 0; } -static void toggle_online_status(bool online) +static int toggle_online_status(bool online) { long val = online ? 1 : 0; int ret; ret = file_write_long(online_path, val); if (ret != 0) - fprintf(stderr, "Failed to bring CPU %s (%s)", + fprintf(stderr, "Failed to bring CPU %s (%s)\n", online ? "online" : "offline", strerror(errno)); + + return ret; } static enum scx_test_status setup(void **ctx) @@ -44,6 +46,7 @@ static enum scx_test_status test_hotplug(bool onlining, bool cbs_defined) { struct hotplug *skel; struct bpf_link *link; + enum scx_test_status status = SCX_TEST_FAIL; long kind, code; SCX_ASSERT(is_cpu_online()); @@ -54,8 +57,8 @@ static enum scx_test_status test_hotplug(bool onlining, bool cbs_defined) SCX_FAIL_IF(hotplug__load(skel), "Failed to load skel"); /* Testing the offline -> online path, so go offline before starting */ - if (onlining) - toggle_online_status(0); + if (onlining && toggle_online_status(0)) + goto out_destroy_skel; if (cbs_defined) { kind = SCX_KIND_VAL(SCX_EXIT_UNREG_BPF); @@ -79,7 +82,8 @@ static enum scx_test_status test_hotplug(bool onlining, bool cbs_defined) return SCX_TEST_FAIL; } - toggle_online_status(onlining ? 1 : 0); + if (toggle_online_status(onlining ? 1 : 0)) + goto out_destroy_link; while (!UEI_EXITED(skel, uei)) sched_yield(); @@ -87,20 +91,23 @@ static enum scx_test_status test_hotplug(bool onlining, bool cbs_defined) SCX_EQ(skel->data->uei.kind, kind); SCX_EQ(UEI_REPORT(skel, uei), code); - if (!onlining) - toggle_online_status(1); + if (!onlining && toggle_online_status(1)) + goto out_destroy_link; + status = SCX_TEST_PASS; +out_destroy_link: bpf_link__destroy(link); +out_destroy_skel: hotplug__destroy(skel); - return SCX_TEST_PASS; + return status; } static enum scx_test_status test_hotplug_attach(void) { struct hotplug *skel; struct bpf_link *link; - enum scx_test_status status = SCX_TEST_PASS; + enum scx_test_status status = SCX_TEST_FAIL; long kind, code; SCX_ASSERT(is_cpu_online()); @@ -115,10 +122,12 @@ static enum scx_test_status test_hotplug_attach(void) * Take the CPU offline to increment the global hotplug seq, which * should cause attach to fail due to us setting the hotplug seq above */ - toggle_online_status(0); + if (toggle_online_status(0)) + goto out_destroy_skel; link = bpf_map__attach_struct_ops(skel->maps.hotplug_nocb_ops); - toggle_online_status(1); + if (toggle_online_status(1)) + goto out_destroy_link; SCX_ASSERT(link); while (!UEI_EXITED(skel, uei)) @@ -130,7 +139,10 @@ static enum scx_test_status test_hotplug_attach(void) SCX_EQ(skel->data->uei.kind, kind); SCX_EQ(UEI_REPORT(skel, uei), code); + status = SCX_TEST_PASS; +out_destroy_link: bpf_link__destroy(link); +out_destroy_skel: hotplug__destroy(skel); return status; diff --git a/tools/testing/selftests/sched_ext/kick.bpf.c b/tools/testing/selftests/sched_ext/kick.bpf.c new file mode 100644 index 000000000000..bd9605d1573d --- /dev/null +++ b/tools/testing/selftests/sched_ext/kick.bpf.c @@ -0,0 +1,164 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + */ +#include <scx/common.bpf.h> + +#include "kick_test.h" + +char _license[] SEC("license") = "GPL"; + +const volatile u32 scenario; +const volatile s32 victim_pid; +const volatile s32 challenger_pid; +const volatile s32 target_cpu; +const volatile s32 lazy_resched_override = -1; + +u32 state; +u64 slice_before; +u64 slice_at_resched; +s32 resched_tif; +u64 nr_wait_callbacks; +bool enq_preempt_issued; + +UEI_DEFINE(uei); + +static bool is_trace_scenario(void) +{ + return scenario <= TICK_EXPIRY; +} + +void BPF_STRUCT_OPS(kick_enqueue, struct task_struct *p, u64 enq_flags) +{ + switch (scenario) { + case ENQ_BOTH: + if (p->pid == victim_pid) { + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_INF, enq_flags); + } else if (p->pid == challenger_pid && state == KICK_STATE_QUEUED) { + enq_preempt_issued = true; + scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL, + enq_flags | SCX_ENQ_PREEMPT | + SCX_ENQ_PREEMPT_LAZY); + } else { + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags); + } + goto kick_last; + case INVALID_KICK_IDLE: + scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_PREEMPT_LAZY | SCX_KICK_IDLE); + break; + case INVALID_KICK_UNKNOWN: + scx_bpf_kick_cpu(scx_bpf_task_cpu(p), 1LLU << 63); + break; + } + + scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags); +kick_last: + if (enq_flags & SCX_ENQ_LAST) + scx_bpf_kick_cpu(target_cpu, SCX_KICK_IDLE); +} + +static void set_lazy_resched_override(struct task_struct *p) +{ + if (p->pid == victim_pid && lazy_resched_override >= 0) + scx_bpf_task_set_lazy_resched(p, lazy_resched_override); +} + +void BPF_STRUCT_OPS(kick_running, struct task_struct *p) +{ + if (!is_trace_scenario() || p->pid != victim_pid) + return; + set_lazy_resched_override(p); + if (__sync_val_compare_and_swap(&state, KICK_STATE_ARMED, KICK_STATE_QUEUED) != + KICK_STATE_ARMED) + return; + + slice_before = p->scx.slice; + + switch (scenario) { + case KICK_IMMEDIATE: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT); + break; + case KICK_LAZY: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY); + break; + case KICK_LAZY_THEN_IMMEDIATE: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY); + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT); + break; + case KICK_IMMEDIATE_THEN_LAZY: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT); + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY); + break; + case KICK_PLAIN_THEN_LAZY: + scx_bpf_kick_cpu(target_cpu, 0); + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY); + break; + case KICK_LAZY_THEN_PLAIN: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY); + scx_bpf_kick_cpu(target_cpu, 0); + break; + case KICK_BOTH: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT | SCX_KICK_PREEMPT_LAZY); + break; + case KICK_LAZY_WAIT: + scx_bpf_kick_cpu(target_cpu, SCX_KICK_PREEMPT_LAZY | SCX_KICK_WAIT); + break; + case ENQ_BOTH: + /* The userspace controller wakes a competing task. */ + break; + case TICK_EXPIRY: + /* no kick: the slice runs out at the tick */ + break; + } +} + +SEC("fexit/__resched_curr") +int BPF_PROG(kick_need_resched, struct rq *rq, int tif) +{ + struct task_struct *task = BPF_CORE_READ(rq, curr); + u64 slice; + + if (!task || BPF_CORE_READ(task, pid) != victim_pid || + BPF_CORE_READ(rq, cpu) != target_cpu || state != KICK_STATE_QUEUED) + return 0; + if (scenario == ENQ_BOTH && !enq_preempt_issued) + return 0; + + slice = BPF_CORE_READ(task, scx.slice); + if (slice) + return 0; + + slice_at_resched = slice; + resched_tif = tif; + state = KICK_STATE_RESCHED; + return 0; +} + +SEC("fentry/kick_sync_wait_bal_cb") +int BPF_PROG(kick_wait_callback, struct rq *rq) +{ + if (scenario == KICK_LAZY_WAIT && BPF_CORE_READ(rq, cpu) == target_cpu) + __sync_fetch_and_add(&nr_wait_callbacks, 1); + return 0; +} + +void BPF_STRUCT_OPS(kick_stopping, struct task_struct *p, bool runnable) +{ + if (p->pid == victim_pid && state == KICK_STATE_RESCHED) + state = KICK_STATE_DONE; +} + +void BPF_STRUCT_OPS(kick_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SEC(".struct_ops.link") +struct sched_ext_ops kick_ops = { + .enqueue = (void *)kick_enqueue, + .running = (void *)kick_running, + .stopping = (void *)kick_stopping, + .exit = (void *)kick_exit, + .flags = SCX_OPS_ENQ_LAST, + .name = "kick", +}; diff --git a/tools/testing/selftests/sched_ext/kick.c b/tools/testing/selftests/sched_ext/kick.c new file mode 100644 index 000000000000..7f01602c75b2 --- /dev/null +++ b/tools/testing/selftests/sched_ext/kick.c @@ -0,0 +1,519 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + */ +#define _GNU_SOURCE +#include <bpf/bpf.h> +#include <errno.h> +#include <linux/sched.h> +#include <sched.h> +#include <signal.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/prctl.h> +#include <sys/wait.h> +#include <unistd.h> +#include <scx/common.h> + +#include "kick.bpf.skel.h" +#include "kick_test.h" +#include "scx_test.h" +#include "util.h" + +#define WAIT_LOOPS 3000 + +struct observation { + u64 slice_before; + u64 slice_at_resched; + u64 nr_wait_callbacks; + s32 resched_tif; +}; + +struct kick_ctx { + cpu_set_t original_mask; + int target_cpu; +}; + +static bool enum_supported(const char *type, const char *name) +{ + u64 value; + + return __COMPAT_read_enum(type, name, &value); +} + +static enum scx_test_status setup_controller(void **ctx_ptr) +{ + struct kick_ctx *ctx; + cpu_set_t controller_mask; + int cpu, controller_cpu = -1; + + ctx = calloc(1, sizeof(*ctx)); + SCX_FAIL_IF(!ctx, "Failed to allocate context"); + if (sched_getaffinity(0, sizeof(ctx->original_mask), &ctx->original_mask)) { + free(ctx); + SCX_FAIL("Failed to get affinity (%d)", errno); + } + + for (cpu = 0; cpu < CPU_SETSIZE; cpu++) { + if (!CPU_ISSET(cpu, &ctx->original_mask)) + continue; + if (controller_cpu < 0) { + controller_cpu = cpu; + } else { + ctx->target_cpu = cpu; + break; + } + } + if (cpu == CPU_SETSIZE) { + printf("SKIP: two allowed CPUs are required\n"); + free(ctx); + return SCX_TEST_SKIP; + } + + CPU_ZERO(&controller_mask); + CPU_SET(controller_cpu, &controller_mask); + if (sched_setaffinity(0, sizeof(controller_mask), &controller_mask)) { + free(ctx); + SCX_FAIL("Failed to pin controller to CPU %d (%d)", controller_cpu, errno); + } + + *ctx_ptr = ctx; + return SCX_TEST_PASS; +} + +static void cleanup_controller(void *ctx_ptr) +{ + struct kick_ctx *ctx = ctx_ptr; + + sched_setaffinity(0, sizeof(ctx->original_mask), &ctx->original_mask); + free(ctx); +} + +static bool wait_for_state(struct kick *skel, u32 wanted) +{ + int i; + + for (i = 0; i < WAIT_LOOPS; i++) { + if (__atomic_load_n(&skel->bss->state, __ATOMIC_ACQUIRE) == wanted) + return true; + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) + return false; + usleep(1000); + } + return false; +} + +static bool wait_for_counter(struct kick *skel, const u64 *counter, u64 wanted) +{ + int i; + + for (i = 0; i < WAIT_LOOPS; i++) { + if (__atomic_load_n(counter, __ATOMIC_ACQUIRE) >= wanted) + return true; + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) + return false; + usleep(1000); + } + return false; +} + +static enum scx_test_status trace_one(struct kick_ctx *ctx, u32 scenario, u64 ops_flags, + s32 lazy_resched_override, struct observation *obs) +{ + struct bpf_link *ops_link = NULL; + struct kick *skel = NULL; + struct scx_test_gated_worker challenger = { .pid = -1, .start_fd = -1 }; + struct scx_test_gated_worker victim; + enum scx_test_status ret = SCX_TEST_FAIL; + int cpu = ctx->target_cpu; + + victim = scx_test_spawn_gated_worker(cpu, false); + if (victim.pid < 0) { + SCX_ERR("Failed to spawn victim"); + return SCX_TEST_FAIL; + } + if (scenario == ENQ_BOTH) { + challenger = scx_test_spawn_gated_worker(cpu, false); + if (challenger.pid < 0) { + SCX_ERR("Failed to spawn enqueue challenger"); + goto out; + } + } + + skel = kick__open(); + if (!skel) { + SCX_ERR("Failed to open scenario %u", scenario); + goto out; + } + SCX_ENUM_INIT(skel); + skel->rodata->scenario = scenario; + skel->rodata->victim_pid = victim.pid; + skel->rodata->challenger_pid = challenger.pid; + skel->rodata->target_cpu = cpu; + skel->rodata->lazy_resched_override = lazy_resched_override; + skel->struct_ops.kick_ops->flags |= ops_flags; + if (scenario != KICK_LAZY_WAIT) + bpf_program__set_autoload(skel->progs.kick_wait_callback, false); + if (kick__load(skel)) { + SCX_ERR("Failed to load scenario %u", scenario); + goto out; + } + + bpf_map__set_autoattach(skel->maps.kick_ops, false); + if (kick__attach(skel)) { + SCX_ERR("Failed to attach __resched_curr tracer"); + goto out; + } + skel->bss->state = KICK_STATE_ARMED; + ops_link = bpf_map__attach_struct_ops(skel->maps.kick_ops); + if (!ops_link) { + SCX_ERR("Failed to attach scenario %u", scenario); + goto out; + } + if (!scx_test_start_gated_worker(&victim)) { + SCX_ERR("Failed to start victim"); + goto out; + } + if (scenario == ENQ_BOTH) { + if (!wait_for_state(skel, KICK_STATE_QUEUED)) { + SCX_ERR("Enqueue scenario did not arm"); + goto out; + } + if (!scx_test_start_gated_worker(&challenger)) { + SCX_ERR("Failed to start enqueue challenger"); + goto out; + } + } + if (!wait_for_state(skel, KICK_STATE_DONE)) { + SCX_ERR("Scenario %u stopped in state %u, exit kind %d", scenario, skel->bss->state, + skel->data->uei.kind); + goto out; + } + if (scenario == KICK_LAZY_WAIT && + !wait_for_counter(skel, &skel->bss->nr_wait_callbacks, 1)) { + SCX_ERR("WAIT callback did not run"); + goto out; + } + + obs->slice_before = skel->bss->slice_before; + obs->slice_at_resched = skel->bss->slice_at_resched; + obs->nr_wait_callbacks = skel->bss->nr_wait_callbacks; + obs->resched_tif = skel->bss->resched_tif; + ret = SCX_TEST_PASS; +out: + scx_test_stop_gated_worker(&victim); + scx_test_stop_gated_worker(&challenger); + if (ops_link) + bpf_link__destroy(ops_link); + if (skel) + kick__destroy(skel); + return ret; +} + +static bool observation_valid(const struct observation *obs) +{ + return obs->slice_before > 0 && !obs->slice_at_resched; +} + +static bool trace_target_supported(const char *name, enum bpf_attach_type attach_type) +{ + return libbpf_find_vmlinux_btf_id(name, attach_type) > 0; +} + +static enum scx_test_status setup_immediate(void **ctx) +{ + if (!enum_supported("scx_kick_flags", "SCX_KICK_PREEMPT")) { + printf("SKIP: SCX_KICK_PREEMPT is not supported\n"); + return SCX_TEST_SKIP; + } + if (!trace_target_supported("__resched_curr", BPF_TRACE_FEXIT)) { + printf("SKIP: __resched_curr is not available in BTF\n"); + return SCX_TEST_SKIP; + } + return setup_controller(ctx); +} + +static enum scx_test_status setup_lazy(void **ctx) +{ + if (!enum_supported("scx_kick_flags", "SCX_KICK_PREEMPT_LAZY")) { + printf("SKIP: SCX_KICK_PREEMPT_LAZY is not supported\n"); + return SCX_TEST_SKIP; + } + return setup_immediate(ctx); +} + +static enum scx_test_status setup_tick(void **ctx) +{ + if (!enum_supported("scx_ops_flags", "SCX_OPS_LAZY_RESCHED")) { + printf("SKIP: SCX_OPS_LAZY_RESCHED is not supported\n"); + return SCX_TEST_SKIP; + } + return setup_lazy(ctx); +} + +static enum scx_test_status setup_coalesce(void **ctx) +{ + if (!enum_supported("scx_enq_flags", "SCX_ENQ_PREEMPT_LAZY")) { + printf("SKIP: SCX_ENQ_PREEMPT_LAZY is not supported\n"); + return SCX_TEST_SKIP; + } + if (!trace_target_supported("kick_sync_wait_bal_cb", BPF_TRACE_FENTRY)) { + printf("SKIP: kick_sync_wait_bal_cb is not available in BTF\n"); + return SCX_TEST_SKIP; + } + return setup_lazy(ctx); +} + +static enum scx_test_status setup_invalid(void **ctx) +{ + if (!enum_supported("scx_kick_flags", "SCX_KICK_PREEMPT_LAZY")) { + printf("SKIP: SCX_KICK_PREEMPT_LAZY is not supported\n"); + return SCX_TEST_SKIP; + } + return setup_controller(ctx); +} + +static enum scx_test_status run_immediate(void *ctx) +{ + struct observation obs; + enum scx_test_status status; + + status = trace_one(ctx, KICK_IMMEDIATE, 0, -1, &obs); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&obs)); + return SCX_TEST_PASS; +} + +static enum scx_test_status run_lazy(void *ctx) +{ + struct observation immediate, lazy; + enum scx_test_status status; + int lazy_mode = scx_test_preempt_lazy_mode(); + + if (lazy_mode < 0) { + printf("SKIP: running kernel preemption mode is unavailable\n"); + return SCX_TEST_SKIP; + } + + status = trace_one(ctx, KICK_IMMEDIATE, 0, -1, &immediate); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&immediate)); + status = trace_one(ctx, KICK_LAZY, 0, -1, &lazy); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&lazy)); + + if (lazy_mode > 0) + SCX_FAIL_IF(lazy.resched_tif == immediate.resched_tif, + "Lazy mode used immediate TIF %d", lazy.resched_tif); + else if (!lazy_mode) + SCX_EQ(lazy.resched_tif, immediate.resched_tif); + + return SCX_TEST_PASS; +} + +static enum scx_test_status run_coalesce(void *ctx) +{ + struct observation immediate, lazy_first, immediate_first; + struct observation plain_first, lazy_then_plain; + struct observation both, lazy_wait, enq_both; + enum scx_test_status status; + + if (scx_test_preempt_lazy_mode() != 1) { + printf("SKIP: kick coalescing requires active lazy preemption\n"); + return SCX_TEST_SKIP; + } + + status = trace_one(ctx, KICK_IMMEDIATE, 0, -1, &immediate); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&immediate)); + status = trace_one(ctx, KICK_LAZY_THEN_IMMEDIATE, 0, -1, &lazy_first); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&lazy_first)); + status = trace_one(ctx, KICK_IMMEDIATE_THEN_LAZY, 0, -1, &immediate_first); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&immediate_first)); + SCX_EQ(lazy_first.resched_tif, immediate.resched_tif); + SCX_EQ(immediate_first.resched_tif, immediate.resched_tif); + + /* + * A plain kick in the same batch doesn't clear the slice by itself. The + * lazy preemption must still expire it, and the plain kick must still + * reschedule immediately, whichever came first. + */ + status = trace_one(ctx, KICK_PLAIN_THEN_LAZY, 0, -1, &plain_first); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&plain_first)); + status = trace_one(ctx, KICK_LAZY_THEN_PLAIN, 0, -1, &lazy_then_plain); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&lazy_then_plain)); + SCX_EQ(plain_first.resched_tif, immediate.resched_tif); + SCX_EQ(lazy_then_plain.resched_tif, immediate.resched_tif); + + /* Immediate kick and WAIT both take precedence over lazy preemption. */ + status = trace_one(ctx, KICK_BOTH, 0, -1, &both); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&both)); + status = trace_one(ctx, KICK_LAZY_WAIT, 0, -1, &lazy_wait); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&lazy_wait)); + SCX_EQ(both.resched_tif, immediate.resched_tif); + SCX_EQ(lazy_wait.resched_tif, immediate.resched_tif); + SCX_GT(lazy_wait.nr_wait_callbacks, 0); + + /* Immediate enqueue preemption likewise takes precedence over lazy. */ + status = trace_one(ctx, ENQ_BOTH, 0, -1, &enq_both); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&enq_both)); + SCX_EQ(enq_both.resched_tif, immediate.resched_tif); + return SCX_TEST_PASS; +} + +/* + * A slice running out at the tick reschedules immediately by default and lazily + * with SCX_OPS_LAZY_RESCHED, the way fair.c expires a slice. + */ +static enum scx_test_status run_tick(void *ctx) +{ + struct observation immediate, lazy, force_lazy, force_immediate; + enum scx_test_status status; + u64 lazy_flag; + int lazy_mode = scx_test_preempt_lazy_mode(); + bool found; + + if (lazy_mode < 0) { + printf("SKIP: running kernel preemption mode is unavailable\n"); + return SCX_TEST_SKIP; + } + + found = __COMPAT_read_enum("scx_ops_flags", "SCX_OPS_LAZY_RESCHED", &lazy_flag); + SCX_ASSERT(found); + status = trace_one(ctx, TICK_EXPIRY, 0, -1, &immediate); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&immediate)); + status = trace_one(ctx, TICK_EXPIRY, lazy_flag, -1, &lazy); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&lazy)); + if (lazy_mode > 0) { + SCX_FAIL_IF(lazy.resched_tif == immediate.resched_tif, + "Lazy slice expiry used immediate TIF %d", lazy.resched_tif); + + status = trace_one(ctx, TICK_EXPIRY, 0, 1, &force_lazy); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&force_lazy)); + status = trace_one(ctx, TICK_EXPIRY, lazy_flag, 0, &force_immediate); + SCX_EQ(status, SCX_TEST_PASS); + SCX_ASSERT(observation_valid(&force_immediate)); + SCX_EQ(force_lazy.resched_tif, lazy.resched_tif); + SCX_EQ(force_immediate.resched_tif, immediate.resched_tif); + } else { + SCX_EQ(lazy.resched_tif, immediate.resched_tif); + } + + return SCX_TEST_PASS; +} + +static enum scx_test_status invalid_one(struct kick_ctx *ctx, u32 scenario) +{ + struct bpf_link *ops_link = NULL; + struct kick *skel = NULL; + struct scx_test_gated_worker victim; + enum scx_test_status ret = SCX_TEST_FAIL; + int cpu = ctx->target_cpu; + int i; + + victim = scx_test_spawn_gated_worker(cpu, false); + if (victim.pid < 0) + return SCX_TEST_FAIL; + + skel = kick__open(); + if (!skel) + goto out; + SCX_ENUM_INIT(skel); + skel->rodata->scenario = scenario; + bpf_program__set_autoload(skel->progs.kick_need_resched, false); + bpf_program__set_autoload(skel->progs.kick_wait_callback, false); + if (kick__load(skel)) + goto out; + ops_link = bpf_map__attach_struct_ops(skel->maps.kick_ops); + if (!ops_link || !scx_test_start_gated_worker(&victim)) + goto out; + + for (i = 0; i < WAIT_LOOPS; i++) { + if (skel->data->uei.kind == EXIT_KIND(SCX_EXIT_ERROR)) { + ret = SCX_TEST_PASS; + break; + } + usleep(1000); + } +out: + scx_test_stop_gated_worker(&victim); + if (ops_link) + bpf_link__destroy(ops_link); + if (skel) + kick__destroy(skel); + return ret; +} + +static enum scx_test_status run_invalid(void *ctx) +{ + enum scx_test_status status; + u32 scenario; + + for (scenario = INVALID_KICK_IDLE; + scenario <= INVALID_KICK_UNKNOWN; scenario++) { + status = invalid_one(ctx, scenario); + SCX_EQ(status, SCX_TEST_PASS); + } + return SCX_TEST_PASS; +} + +static struct scx_test kick_immediate = { + .name = "kick_immediate", + .description = "Trace immediate kick slice expiration and rescheduling", + .setup = setup_immediate, + .run = run_immediate, + .cleanup = cleanup_controller, +}; + +static struct scx_test kick_lazy = { + .name = "kick_lazy", + .description = "Trace lazy kick slice expiration and rescheduling", + .setup = setup_lazy, + .run = run_lazy, + .cleanup = cleanup_controller, +}; + +static struct scx_test kick_coalesce = { + .name = "kick_coalesce", + .description = "Verify lazy preemption coalesces with immediate requests", + .setup = setup_coalesce, + .run = run_coalesce, + .cleanup = cleanup_controller, +}; + +static struct scx_test kick_tick = { + .name = "kick_tick", + .description = "Trace slice expiry at the tick, immediate and lazy", + .setup = setup_tick, + .run = run_tick, + .cleanup = cleanup_controller, +}; + +static struct scx_test kick_invalid = { + .name = "kick_invalid", + .description = "Verify invalid kick flag combinations fail", + .setup = setup_invalid, + .run = run_invalid, + .cleanup = cleanup_controller, +}; + +__attribute__((constructor)) +static void register_kick_tests(void) +{ + scx_test_register(&kick_immediate); + scx_test_register(&kick_lazy); + scx_test_register(&kick_coalesce); + scx_test_register(&kick_tick); + scx_test_register(&kick_invalid); +} diff --git a/tools/testing/selftests/sched_ext/kick_test.h b/tools/testing/selftests/sched_ext/kick_test.h new file mode 100644 index 000000000000..7d3c608c6b63 --- /dev/null +++ b/tools/testing/selftests/sched_ext/kick_test.h @@ -0,0 +1,29 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES */ +#ifndef __KICK_TEST_H__ +#define __KICK_TEST_H__ + +enum kick_scenario { + KICK_IMMEDIATE, + KICK_LAZY, + KICK_LAZY_THEN_IMMEDIATE, + KICK_IMMEDIATE_THEN_LAZY, + KICK_PLAIN_THEN_LAZY, + KICK_LAZY_THEN_PLAIN, + KICK_BOTH, + KICK_LAZY_WAIT, + ENQ_BOTH, + TICK_EXPIRY, + INVALID_KICK_IDLE, + INVALID_KICK_UNKNOWN, +}; + +enum kick_state { + KICK_STATE_IDLE, + KICK_STATE_ARMED, + KICK_STATE_QUEUED, + KICK_STATE_RESCHED, + KICK_STATE_DONE, +}; + +#endif /* __KICK_TEST_H__ */ diff --git a/tools/testing/selftests/sched_ext/nohz_tick.bpf.c b/tools/testing/selftests/sched_ext/nohz_tick.bpf.c index 6998c5dd6bcb..5fec5673fbf2 100644 --- a/tools/testing/selftests/sched_ext/nohz_tick.bpf.c +++ b/tools/testing/selftests/sched_ext/nohz_tick.bpf.c @@ -6,13 +6,22 @@ */ #include <scx/common.bpf.h> +#include "nohz_tick_test.h" + char _license[] SEC("license") = "GPL"; const volatile s32 test_cpu; -bool finite_phase; +u32 phase; +s32 victim_pid; +s32 challenger_pid; +s32 trigger_pid; u64 nr_inf_running; u64 nr_finite_running; u64 nr_finite_ticks; +u64 nr_lazy_victim_running; +u64 nr_lazy_enq_running; +u64 nr_lazy_kick_running; +u64 nr_lazy_ticks; UEI_DEFINE(uei); @@ -24,9 +33,31 @@ s32 BPF_STRUCT_OPS(nohz_tick_select_cpu, struct task_struct *p, s32 prev_cpu, void BPF_STRUCT_OPS(nohz_tick_enqueue, struct task_struct *p, u64 enq_flags) { - u64 slice = finite_phase ? 1000000ULL : SCX_SLICE_INF; + u64 slice; + u64 dsq_id = SCX_DSQ_GLOBAL; + + switch (phase) { + case NOHZ_PHASE_INF: + slice = SCX_SLICE_INF; + break; + case NOHZ_PHASE_FINITE: + slice = 1000000ULL; + break; + case NOHZ_PHASE_LAZY_ENQ: + case NOHZ_PHASE_LAZY_KICK: + dsq_id = SCX_DSQ_LOCAL; + slice = p->pid == victim_pid ? SCX_SLICE_INF : SCX_SLICE_DFL; + if (phase == NOHZ_PHASE_LAZY_ENQ && p->pid == challenger_pid) + enq_flags |= SCX_ENQ_PREEMPT_LAZY; + break; + default: + slice = SCX_SLICE_DFL; + break; + } - scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, slice, enq_flags); + scx_bpf_dsq_insert(p, dsq_id, slice, enq_flags); + if (phase == NOHZ_PHASE_LAZY_KICK && p->pid == trigger_pid) + scx_bpf_kick_cpu(test_cpu, SCX_KICK_PREEMPT_LAZY); if (enq_flags & SCX_ENQ_LAST) scx_bpf_kick_cpu(test_cpu, SCX_KICK_IDLE); } @@ -36,16 +67,28 @@ void BPF_STRUCT_OPS(nohz_tick_running, struct task_struct *p) if (bpf_get_smp_processor_id() != test_cpu) return; - if (finite_phase) + if (phase == NOHZ_PHASE_FINITE) __sync_fetch_and_add(&nr_finite_running, 1); - else + else if (phase == NOHZ_PHASE_INF) __sync_fetch_and_add(&nr_inf_running, 1); + else if (p->pid == victim_pid) + __sync_fetch_and_add(&nr_lazy_victim_running, 1); + else if (phase == NOHZ_PHASE_LAZY_ENQ && p->pid == challenger_pid) + __sync_fetch_and_add(&nr_lazy_enq_running, 1); + else if (phase == NOHZ_PHASE_LAZY_KICK && p->pid == challenger_pid) + __sync_fetch_and_add(&nr_lazy_kick_running, 1); } void BPF_STRUCT_OPS(nohz_tick_tick, struct task_struct *p) { - if (bpf_get_smp_processor_id() == test_cpu && finite_phase) + if (bpf_get_smp_processor_id() != test_cpu) + return; + + if (phase == NOHZ_PHASE_FINITE) __sync_fetch_and_add(&nr_finite_ticks, 1); + else if ((phase == NOHZ_PHASE_LAZY_ENQ || + phase == NOHZ_PHASE_LAZY_KICK) && p->pid == victim_pid) + __sync_fetch_and_add(&nr_lazy_ticks, 1); } void BPF_STRUCT_OPS(nohz_tick_exit, struct scx_exit_info *ei) @@ -61,5 +104,5 @@ struct sched_ext_ops nohz_tick_ops = { .tick = (void *)nohz_tick_tick, .exit = (void *)nohz_tick_exit, .name = "nohz_tick", - .timeout_ms = 1000U, + .timeout_ms = 5000U, }; diff --git a/tools/testing/selftests/sched_ext/nohz_tick.c b/tools/testing/selftests/sched_ext/nohz_tick.c index 028f54391c2c..df40976b9166 100644 --- a/tools/testing/selftests/sched_ext/nohz_tick.c +++ b/tools/testing/selftests/sched_ext/nohz_tick.c @@ -21,19 +21,24 @@ #include <scx/common.h> #include "nohz_tick.bpf.skel.h" +#include "nohz_tick_test.h" #include "scx_test.h" +#include "util.h" #ifndef SCHED_EXT #define SCHED_EXT 7 #endif #define MIN_FINITE_TICKS 3 -#define PHASE_TIMEOUT_MS 1000 +#define PHASE_TIMEOUT_MS 5000 +#define TICK_STOP_STABLE_MS 100 struct nohz_tick_ctx { struct nohz_tick *skel; cpu_set_t original_mask; int test_cpu; + int housekeeping_cpu; + bool test_lazy; }; static int first_allowed_cpu(const cpu_set_t *mask, int first, int last) @@ -47,20 +52,21 @@ static int first_allowed_cpu(const cpu_set_t *mask, int first, int last) return -1; } -static int find_nohz_full_cpu(const cpu_set_t *allowed) +static int read_nohz_full_mask(cpu_set_t *mask) { char buf[4096], *cur, *end; FILE *file; + int ret = 0; file = fopen("/sys/devices/system/cpu/nohz_full", "r"); if (!file) - return -1; + return -errno; if (!fgets(buf, sizeof(buf), file)) { - fclose(file); - return -1; + ret = ferror(file) ? -errno : -EINVAL; + goto out; } - fclose(file); + CPU_ZERO(mask); cur = buf; while (*cur) { long first, last; @@ -73,25 +79,30 @@ static int find_nohz_full_cpu(const cpu_set_t *allowed) errno = 0; first = strtol(cur, &end, 10); - if (errno || end == cur || first < 0 || first >= CPU_SETSIZE) - return -1; + if (errno || end == cur || first < 0 || first >= CPU_SETSIZE) { + ret = -EINVAL; + goto out; + } cur = end; last = first; if (*cur == '-') { cur++; errno = 0; last = strtol(cur, &end, 10); - if (errno || end == cur || last < first) - return -1; + if (errno || end == cur || last < first) { + ret = -EINVAL; + goto out; + } cur = end; } - cpu = first_allowed_cpu(allowed, first, last); - if (cpu >= 0) - return cpu; + for (cpu = first; cpu <= last && cpu < CPU_SETSIZE; cpu++) + CPU_SET(cpu, mask); } - return -1; +out: + fclose(file); + return ret; } static pid_t start_worker(int cpu) @@ -146,14 +157,41 @@ static int pause_worker(pid_t pid) return 0; } -static bool wait_for_counter(const u64 *counter, u64 value, int timeout_ms) +static bool wait_for_counter(struct nohz_tick *skel, const u64 *counter, u64 value, + int timeout_ms) { int elapsed; for (elapsed = 0; elapsed < timeout_ms; elapsed++) { if (__atomic_load_n(counter, __ATOMIC_RELAXED) >= value) return true; + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) + return false; + usleep(1000); + } + + return false; +} + +static bool wait_for_tick_stop(struct nohz_tick *skel, const u64 *counter, int timeout_ms) +{ + u64 prev = __atomic_load_n(counter, __ATOMIC_RELAXED); + int elapsed, stable = 0; + + for (elapsed = 0; elapsed < timeout_ms; elapsed++) { + u64 curr; + usleep(1000); + if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) + return false; + curr = __atomic_load_n(counter, __ATOMIC_RELAXED); + if (curr == prev) { + if (++stable >= TICK_STOP_STABLE_MS) + return true; + } else { + prev = curr; + stable = 0; + } } return false; @@ -162,8 +200,10 @@ static bool wait_for_counter(const u64 *counter, u64 value, int timeout_ms) static enum scx_test_status setup(void **ctx_ptr) { struct nohz_tick_ctx *ctx; - cpu_set_t controller_mask; - int cpu; + cpu_set_t controller_mask, nohz_full_mask; + bool lazy_supported; + u64 enum_value, lazy_ops_flag; + int cpu, i, lazy_mode, ret; ctx = calloc(1, sizeof(*ctx)); SCX_FAIL_IF(!ctx, "Failed to allocate context"); @@ -173,15 +213,26 @@ static enum scx_test_status setup(void **ctx_ptr) SCX_FAIL("Failed to get affinity (%d)", errno); } - cpu = find_nohz_full_cpu(&ctx->original_mask); - if (cpu < 0) { + ret = read_nohz_full_mask(&nohz_full_mask); + if (ret) { + fprintf(stderr, "SKIP: failed to read NOHZ_FULL mask (%d)\n", ret); + free(ctx); + return SCX_TEST_SKIP; + } + + for (cpu = 0; cpu < CPU_SETSIZE; cpu++) + if (CPU_ISSET(cpu, &ctx->original_mask) && CPU_ISSET(cpu, &nohz_full_mask)) + break; + if (cpu == CPU_SETSIZE) { fprintf(stderr, "SKIP: no allowed NOHZ_FULL CPU\n"); free(ctx); return SCX_TEST_SKIP; } controller_mask = ctx->original_mask; - CPU_CLR(cpu, &controller_mask); + for (i = 0; i < CPU_SETSIZE; i++) + if (CPU_ISSET(i, &nohz_full_mask)) + CPU_CLR(i, &controller_mask); if (CPU_COUNT(&controller_mask) == 0) { fprintf(stderr, "SKIP: no housekeeping CPU available\n"); free(ctx); @@ -189,6 +240,21 @@ static enum scx_test_status setup(void **ctx_ptr) } ctx->test_cpu = cpu; + ctx->housekeeping_cpu = first_allowed_cpu(&controller_mask, 0, CPU_SETSIZE - 1); + lazy_supported = __COMPAT_read_enum("scx_enq_flags", "SCX_ENQ_PREEMPT_LAZY", + &enum_value) && + __COMPAT_read_enum("scx_kick_flags", "SCX_KICK_PREEMPT_LAZY", + &enum_value) && + __COMPAT_read_enum("scx_ops_flags", "SCX_OPS_LAZY_RESCHED", + &lazy_ops_flag); + lazy_mode = scx_test_preempt_lazy_mode(); + ctx->test_lazy = lazy_supported && lazy_mode > 0; + if (lazy_supported && lazy_mode < 0) + fprintf(stderr, + "SKIP: kernel preemption mode unavailable; skipping lazy NOHZ_FULL phases\n"); + else if (lazy_supported && !lazy_mode) + fprintf(stderr, + "SKIP: lazy preemption inactive; skipping lazy NOHZ_FULL phases\n"); ctx->skel = nohz_tick__open(); if (!ctx->skel) { free(ctx); @@ -199,6 +265,8 @@ static enum scx_test_status setup(void **ctx_ptr) ctx->skel->rodata->test_cpu = cpu; ctx->skel->struct_ops.nohz_tick_ops->flags |= SCX_OPS_SWITCH_PARTIAL | SCX_OPS_ENQ_LAST; + if (lazy_supported) + ctx->skel->struct_ops.nohz_tick_ops->flags |= lazy_ops_flag; if (nohz_tick__load(ctx->skel)) { nohz_tick__destroy(ctx->skel); free(ctx); @@ -220,11 +288,15 @@ static enum scx_test_status run(void *ctx_ptr) struct nohz_tick_ctx *ctx = ctx_ptr; struct nohz_tick *skel = ctx->skel; struct bpf_link *link = NULL; + struct scx_test_gated_worker victim = { .pid = -1, .start_fd = -1 }; + struct scx_test_gated_worker challenger = { .pid = -1, .start_fd = -1 }; + struct scx_test_gated_worker trigger = { .pid = -1, .start_fd = -1 }; enum scx_test_status status = SCX_TEST_FAIL; pid_t finite_worker = -1; pid_t inf_worker = -1; u64 finite_running; u64 finite_ticks; + u64 victim_running; int ret; link = bpf_map__attach_struct_ops(skel->maps.nohz_tick_ops); @@ -241,7 +313,7 @@ static enum scx_test_status run(void *ctx_ptr) SCX_ERR("Failed to start infinite-slice worker (%d)", errno); goto out; } - if (!wait_for_counter(&skel->bss->nr_inf_running, 1, + if (!wait_for_counter(skel, &skel->bss->nr_inf_running, 1, PHASE_TIMEOUT_MS)) { SCX_ERR("Infinite-slice worker was not scheduled"); goto out; @@ -260,18 +332,18 @@ static enum scx_test_status run(void *ctx_ptr) /* * The next EXT task receives a finite slice and must restart the tick. */ - __atomic_store_n(&skel->bss->finite_phase, true, __ATOMIC_RELEASE); + __atomic_store_n(&skel->bss->phase, NOHZ_PHASE_FINITE, __ATOMIC_RELEASE); finite_worker = start_worker(ctx->test_cpu); if (finite_worker < 0) { SCX_ERR("Failed to start finite-slice worker (%d)", errno); goto out; } - if (!wait_for_counter(&skel->bss->nr_finite_running, 1, + if (!wait_for_counter(skel, &skel->bss->nr_finite_running, 1, PHASE_TIMEOUT_MS)) { SCX_ERR("Finite-slice worker was not scheduled"); goto out; } - if (!wait_for_counter(&skel->bss->nr_finite_ticks, MIN_FINITE_TICKS, + if (!wait_for_counter(skel, &skel->bss->nr_finite_ticks, MIN_FINITE_TICKS, PHASE_TIMEOUT_MS)) { SCX_ERR("Finite-slice worker received only %llu scheduler ticks", (unsigned long long)skel->bss->nr_finite_ticks); @@ -295,12 +367,12 @@ static enum scx_test_status run(void *ctx_ptr) SCX_ERR("Failed to start second finite-slice worker (%d)", errno); goto out; } - if (!wait_for_counter(&skel->bss->nr_finite_running, + if (!wait_for_counter(skel, &skel->bss->nr_finite_running, finite_running + 1, PHASE_TIMEOUT_MS)) { SCX_ERR("Second finite-slice worker was not scheduled"); goto out; } - if (!wait_for_counter(&skel->bss->nr_finite_ticks, + if (!wait_for_counter(skel, &skel->bss->nr_finite_ticks, finite_ticks + MIN_FINITE_TICKS, PHASE_TIMEOUT_MS)) { SCX_ERR("Second finite-slice worker received only %llu scheduler ticks", @@ -308,7 +380,87 @@ static enum scx_test_status run(void *ctx_ptr) finite_ticks)); goto out; } + stop_worker(finite_worker); + finite_worker = -1; + stop_worker(inf_worker); + inf_worker = -1; + if (!ctx->test_lazy) + goto check_exit; + + /* + * A lazy local enqueue must restart the tick after clearing the slice + * of an infinite-slice task on a full-dynticks CPU. + */ + __atomic_store_n(&skel->bss->phase, NOHZ_PHASE_LAZY_ENQ, __ATOMIC_RELEASE); + victim = scx_test_spawn_gated_worker(ctx->test_cpu, true); + challenger = scx_test_spawn_gated_worker(ctx->test_cpu, true); + if (victim.pid < 0 || challenger.pid < 0) { + SCX_ERR("Failed to spawn lazy-enqueue workers"); + goto out; + } + skel->bss->victim_pid = victim.pid; + skel->bss->challenger_pid = challenger.pid; + if (!scx_test_start_gated_worker(&victim) || + !wait_for_counter(skel, &skel->bss->nr_lazy_victim_running, 1, + PHASE_TIMEOUT_MS)) { + SCX_ERR("Lazy-enqueue victim was not scheduled"); + goto out; + } + if (!wait_for_tick_stop(skel, &skel->bss->nr_lazy_ticks, PHASE_TIMEOUT_MS)) { + SCX_ERR("Tick did not stop before lazy enqueue"); + goto out; + } + + if (!scx_test_start_gated_worker(&challenger) || + !wait_for_counter(skel, &skel->bss->nr_lazy_enq_running, 1, PHASE_TIMEOUT_MS)) { + SCX_ERR("Lazy enqueue made no progress on CPU %d", ctx->test_cpu); + goto out; + } + scx_test_stop_gated_worker(&victim); + scx_test_stop_gated_worker(&challenger); + + /* Repeat with a lazy kick delivered from a housekeeping CPU. */ + __atomic_store_n(&skel->bss->phase, NOHZ_PHASE_LAZY_KICK, __ATOMIC_RELEASE); + victim = scx_test_spawn_gated_worker(ctx->test_cpu, true); + challenger = scx_test_spawn_gated_worker(ctx->test_cpu, true); + trigger = scx_test_spawn_gated_worker(ctx->housekeeping_cpu, true); + if (victim.pid < 0 || challenger.pid < 0 || trigger.pid < 0) { + SCX_ERR("Failed to spawn lazy-kick workers"); + goto out; + } + skel->bss->victim_pid = victim.pid; + skel->bss->challenger_pid = challenger.pid; + skel->bss->trigger_pid = trigger.pid; + victim_running = __atomic_load_n(&skel->bss->nr_lazy_victim_running, + __ATOMIC_RELAXED); + if (!scx_test_start_gated_worker(&victim) || + !wait_for_counter(skel, &skel->bss->nr_lazy_victim_running, victim_running + 1, + PHASE_TIMEOUT_MS)) { + SCX_ERR("Lazy-kick victim was not scheduled"); + goto out; + } + if (!scx_test_start_gated_worker(&challenger)) { + SCX_ERR("Failed to start lazy-kick challenger"); + goto out; + } + if (!wait_for_tick_stop(skel, &skel->bss->nr_lazy_ticks, PHASE_TIMEOUT_MS)) { + SCX_ERR("Tick did not stop before lazy kick"); + goto out; + } + if (__atomic_load_n(&skel->bss->nr_lazy_kick_running, __ATOMIC_RELAXED)) { + SCX_ERR("Lazy-kick challenger ran before the kick"); + goto out; + } + if (!scx_test_start_gated_worker(&trigger) || + !wait_for_counter(skel, &skel->bss->nr_lazy_kick_running, 1, PHASE_TIMEOUT_MS)) { + SCX_ERR("Lazy kick made no progress on CPU %d", ctx->test_cpu); + goto out; + } + scx_test_stop_gated_worker(&trigger); + scx_test_stop_gated_worker(&victim); + scx_test_stop_gated_worker(&challenger); +check_exit: if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) { SCX_ERR("Scheduler exited unexpectedly (kind=%llu code=%lld)", (unsigned long long)skel->data->uei.kind, @@ -321,6 +473,9 @@ static enum scx_test_status run(void *ctx_ptr) (unsigned long long)skel->bss->nr_finite_ticks); status = SCX_TEST_PASS; out: + scx_test_stop_gated_worker(&trigger); + scx_test_stop_gated_worker(&victim); + scx_test_stop_gated_worker(&challenger); stop_worker(finite_worker); stop_worker(inf_worker); if (link) diff --git a/tools/testing/selftests/sched_ext/nohz_tick_test.h b/tools/testing/selftests/sched_ext/nohz_tick_test.h new file mode 100644 index 000000000000..b122d723cd8b --- /dev/null +++ b/tools/testing/selftests/sched_ext/nohz_tick_test.h @@ -0,0 +1,13 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES */ +#ifndef __NOHZ_TICK_TEST_H__ +#define __NOHZ_TICK_TEST_H__ + +enum nohz_phase { + NOHZ_PHASE_INF, + NOHZ_PHASE_FINITE, + NOHZ_PHASE_LAZY_ENQ, + NOHZ_PHASE_LAZY_KICK, +}; + +#endif /* __NOHZ_TICK_TEST_H__ */ diff --git a/tools/testing/selftests/sched_ext/rt_stall.c b/tools/testing/selftests/sched_ext/rt_stall.c index a5041fc2e44f..a1552b38a6e8 100644 --- a/tools/testing/selftests/sched_ext/rt_stall.c +++ b/tools/testing/selftests/sched_ext/rt_stall.c @@ -17,7 +17,6 @@ #include <scx/common.h> #include "rt_stall.bpf.skel.h" #include "scx_test.h" -#include "../kselftest.h" #define CORE_ID 0 /* CPU to pin tasks to */ #define RUN_TIME 5 /* How long to run the test in seconds */ @@ -35,15 +34,17 @@ static void signal_ready(int fd) } /* Wait for a child to signal readiness via a pipe */ -static void wait_ready(int fd) +static bool wait_ready(int fd) { + bool ready; char c; - if (read(fd, &c, 1) != 1) { + ready = read(fd, &c, 1) == 1; + if (!ready) perror("read from ready pipe"); - exit(EXIT_FAILURE); - } close(fd); + + return ready; } /* Simple busy-wait function for test tasks */ @@ -151,13 +152,11 @@ static bool sched_stress_test(bool is_ext) float ext_runtime, rt_runtime, actual_ratio; int ext_pid, rt_pid; int ext_ready[2], rt_ready[2]; - - ksft_print_header(); - ksft_set_plan(1); + bool ret = false; if (pipe(ext_ready) || pipe(rt_ready)) { perror("pipe"); - ksft_exit_fail(); + return false; } /* Create and set up a EXT task */ @@ -172,7 +171,7 @@ static bool sched_stress_test(bool is_ext) exit(0); } else if (ext_pid < 0) { perror("fork task"); - ksft_exit_fail(); + return false; } /* Create an RT task */ @@ -188,7 +187,7 @@ static bool sched_stress_test(bool is_ext) exit(0); } else if (rt_pid < 0) { perror("fork for RT task"); - ksft_exit_fail(); + goto out_kill_ext; } /* @@ -199,45 +198,47 @@ static bool sched_stress_test(bool is_ext) */ close(ext_ready[1]); close(rt_ready[1]); - wait_ready(ext_ready[0]); - wait_ready(rt_ready[0]); + if (!wait_ready(ext_ready[0]) || !wait_ready(rt_ready[0])) + goto out_kill; /* Let the processes run for the specified time */ sleep(RUN_TIME); /* Get runtime for the EXT task */ ext_runtime = get_process_runtime(ext_pid); - if (ext_runtime == -1) - ksft_exit_fail_msg("Error getting runtime for %s task (PID %d)\n", - class_str, ext_pid); - ksft_print_msg("Runtime of %s task (PID %d) is %f seconds\n", - class_str, ext_pid, ext_runtime); + if (ext_runtime == -1) { + fprintf(stderr, "Failed to read %s task runtime\n", class_str); + goto out_kill; + } /* Get runtime for the RT task */ rt_runtime = get_process_runtime(rt_pid); - if (rt_runtime == -1) - ksft_exit_fail_msg("Error getting runtime for RT task (PID %d)\n", rt_pid); - ksft_print_msg("Runtime of RT task (PID %d) is %f seconds\n", rt_pid, rt_runtime); - - /* Kill the processes */ - kill(ext_pid, SIGKILL); - kill(rt_pid, SIGKILL); - waitpid(ext_pid, NULL, 0); - waitpid(rt_pid, NULL, 0); + if (rt_runtime == -1) { + fprintf(stderr, "Failed to read RT task runtime\n"); + goto out_kill; + } /* Verify that the scx task got enough runtime */ actual_ratio = ext_runtime / (ext_runtime + rt_runtime); - ksft_print_msg("%s task got %.2f%% of total runtime\n", - class_str, actual_ratio * 100); + fprintf(stderr, "%s task ran %.3fs, RT task ran %.3fs (%.2f%% of runtime)\n", + class_str, ext_runtime, rt_runtime, actual_ratio * 100); - if (actual_ratio >= expected_min_ratio) { - ksft_test_result_pass("PASS: %s task got more than %.2f%% of runtime\n", - class_str, expected_min_ratio * 100); - return true; + if (actual_ratio < expected_min_ratio) { + fprintf(stderr, "%s task got less than %.2f%% of runtime\n", + class_str, expected_min_ratio * 100); + goto out_kill; } - ksft_test_result_fail("FAIL: %s task got less than %.2f%% of runtime\n", - class_str, expected_min_ratio * 100); - return false; + + ret = true; + +out_kill: + kill(rt_pid, SIGKILL); + waitpid(rt_pid, NULL, 0); +out_kill_ext: + kill(ext_pid, SIGKILL); + waitpid(ext_pid, NULL, 0); + + return ret; } static enum scx_test_status run(void *ctx) @@ -263,14 +264,17 @@ static enum scx_test_status run(void *ctx) link = bpf_map__attach_struct_ops(skel->maps.rt_stall_ops); SCX_FAIL_IF(!link, "Failed to attach scheduler"); } + res = sched_stress_test(is_ext); + if (is_ext) { - SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_NONE)); + int exit_kind = skel->data->uei.kind; bpf_link__destroy(link); + SCX_EQ(exit_kind, EXIT_KIND(SCX_EXIT_NONE)); } if (!res) - ksft_exit_fail(); + return SCX_TEST_FAIL; } return SCX_TEST_PASS; diff --git a/tools/testing/selftests/sched_ext/runner.c b/tools/testing/selftests/sched_ext/runner.c index c264807caa91..57dda75384f1 100644 --- a/tools/testing/selftests/sched_ext/runner.c +++ b/tools/testing/selftests/sched_ext/runner.c @@ -243,7 +243,7 @@ int main(int argc, char **argv) printf(" - %s\n", failed_tests[i]); } - return failed > 0 ? 1 : 0; + return failed > 0 || exit_req ? 1 : 0; } void scx_test_register(struct scx_test *test) diff --git a/tools/testing/selftests/sched_ext/test_modules/Makefile b/tools/testing/selftests/sched_ext/test_modules/Makefile new file mode 100644 index 000000000000..a0e9e9401ead --- /dev/null +++ b/tools/testing/selftests/sched_ext/test_modules/Makefile @@ -0,0 +1,13 @@ +# SPDX-License-Identifier: GPL-2.0 +# Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + +TESTMODS_DIR := $(realpath $(dir $(abspath $(lastword $(MAKEFILE_LIST))))) +KDIR ?= $(if $(O),$(O),$(realpath ../../../../..)) + +obj-m += scx_enq_blocked_test.o + +all: + +$(Q)$(MAKE) -C $(KDIR) M=$(TESTMODS_DIR) modules + +clean: + +$(Q)$(MAKE) -C $(KDIR) M=$(TESTMODS_DIR) clean diff --git a/tools/testing/selftests/sched_ext/test_modules/scx_enq_blocked_test.c b/tools/testing/selftests/sched_ext/test_modules/scx_enq_blocked_test.c new file mode 100644 index 000000000000..908689ed5578 --- /dev/null +++ b/tools/testing/selftests/sched_ext/test_modules/scx_enq_blocked_test.c @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES + * + * Kernel mutex fixture for the sched_ext SCX_OPS_ENQ_BLOCKED selftest. + */ + +#include <linux/atomic.h> +#include <linux/fs.h> +#include <linux/jiffies.h> +#include <linux/ktime.h> +#include <linux/miscdevice.h> +#include <linux/module.h> +#include <linux/mutex.h> +#include <linux/sched.h> +#include <linux/uaccess.h> + +#include "../enq_blocked.h" + +#define DONOR_WAIT_TIMEOUT msecs_to_jiffies(2000) +#define ATTACH_WAIT_TIMEOUT msecs_to_jiffies(10000) +#define MUTEX_HOLD_TIME msecs_to_jiffies(200) + +static DEFINE_MUTEX(test_mutex); +static DEFINE_SPINLOCK(donor_lock); +static struct task_struct *donor_task; +static atomic_t owner_ready = ATOMIC_INIT(0); +static atomic_t donor_started = ATOMIC_INIT(0); +static atomic_t attach_pending = ATOMIC_INIT(0); +static atomic_t attach_done = ATOMIC_INIT(0); +static atomic64_t hold_time_ns = ATOMIC64_INIT(0); +static atomic64_t wait_time_ns = ATOMIC64_INIT(0); +static atomic64_t nr_holds = ATOMIC64_INIT(0); +static atomic64_t nr_waits = ATOMIC64_INIT(0); + +static long run_owner(void) +{ + unsigned long timeout; + u64 start_ns; + long ret = 0; + + atomic_set(&donor_started, 0); + mutex_lock(&test_mutex); + start_ns = ktime_get_ns(); + atomic_set(&owner_ready, 1); + + timeout = jiffies + DONOR_WAIT_TIMEOUT; + while (!atomic_read(&donor_started)) { + if (time_after(jiffies, timeout)) { + ret = -ETIMEDOUT; + goto out; + } + cond_resched(); + } + if (atomic_xchg(&attach_pending, 0)) { + timeout = jiffies + ATTACH_WAIT_TIMEOUT; + while (!atomic_read(&attach_done)) { + if (time_after(jiffies, timeout)) { + ret = -ETIMEDOUT; + goto out; + } + cond_resched(); + } + } + + /* Keep yielding while the donor blocks on test_mutex. */ + timeout = jiffies + MUTEX_HOLD_TIME; + while (time_before(jiffies, timeout)) + cond_resched(); + +out: + atomic_set(&owner_ready, 0); + atomic64_add(ktime_get_ns() - start_ns, &hold_time_ns); + atomic64_inc(&nr_holds); + mutex_unlock(&test_mutex); + return ret; +} + +static long run_donor(void) +{ + unsigned long flags; + u64 start_ns; + + if (!atomic_read(&owner_ready)) + return -EAGAIN; + + get_task_struct(current); + spin_lock_irqsave(&donor_lock, flags); + WARN_ON_ONCE(donor_task); + donor_task = current; + spin_unlock_irqrestore(&donor_lock, flags); + + atomic_set(&donor_started, 1); + start_ns = ktime_get_ns(); + mutex_lock(&test_mutex); + + spin_lock_irqsave(&donor_lock, flags); + donor_task = NULL; + spin_unlock_irqrestore(&donor_lock, flags); + put_task_struct(current); + + atomic64_add(ktime_get_ns() - start_ns, &wait_time_ns); + atomic64_inc(&nr_waits); + mutex_unlock(&test_mutex); + return 0; +} + +static long get_donor_state(void) +{ + struct task_struct *task; + unsigned long flags; + long state = 0; + + spin_lock_irqsave(&donor_lock, flags); + task = donor_task; + if (task) + get_task_struct(task); + spin_unlock_irqrestore(&donor_lock, flags); + if (!task) + return -ENOENT; + + if (READ_ONCE(task->__state) != TASK_RUNNING) + state |= ENQ_BLOCKED_DONOR_SLEEPING; + if (READ_ONCE(task->on_rq)) + state |= ENQ_BLOCKED_DONOR_ON_RQ; + put_task_struct(task); + return state; +} + +static void reset_stats(void) +{ + atomic64_set(&hold_time_ns, 0); + atomic64_set(&wait_time_ns, 0); + atomic64_set(&nr_holds, 0); + atomic64_set(&nr_waits, 0); +} + +static long get_stats(unsigned long arg) +{ + struct enq_blocked_stats stats = { + .hold_time_ns = atomic64_read(&hold_time_ns), + .wait_time_ns = atomic64_read(&wait_time_ns), + .nr_holds = atomic64_read(&nr_holds), + .nr_waits = atomic64_read(&nr_waits), + }; + + return copy_to_user((void __user *)arg, &stats, sizeof(stats)) ? + -EFAULT : 0; +} + +static long enq_blocked_ioctl(struct file *file, unsigned int cmd, + unsigned long arg) +{ + switch (cmd) { + case ENQ_BLOCKED_IOCTL_OWNER: + return run_owner(); + case ENQ_BLOCKED_IOCTL_DONOR: + return run_donor(); + case ENQ_BLOCKED_IOCTL_RESET_STATS: + reset_stats(); + return 0; + case ENQ_BLOCKED_IOCTL_GET_STATS: + return get_stats(arg); + case ENQ_BLOCKED_IOCTL_PREP_ATTACH: + atomic_set(&attach_done, 0); + atomic_set(&attach_pending, 1); + return 0; + case ENQ_BLOCKED_IOCTL_ATTACH_DONE: + atomic_set(&attach_done, 1); + return 0; + case ENQ_BLOCKED_IOCTL_DONOR_STATE: + return get_donor_state(); + case ENQ_BLOCKED_IOCTL_PROXY_SUPPORTED: + return IS_ENABLED(CONFIG_SCHED_PROXY_EXEC); + default: + return -EINVAL; + } +} + +static const struct file_operations enq_blocked_fops = { + .owner = THIS_MODULE, + .unlocked_ioctl = enq_blocked_ioctl, +}; + +static struct miscdevice enq_blocked_device = { + .minor = MISC_DYNAMIC_MINOR, + .name = "scx_enq_blocked", + .fops = &enq_blocked_fops, + .mode = 0600, +}; + +module_misc_device(enq_blocked_device); +MODULE_AUTHOR("Andrea Righi <arighi@nvidia.com>"); +MODULE_LICENSE("GPL"); +MODULE_DESCRIPTION("sched_ext blocked donor test module"); diff --git a/tools/testing/selftests/sched_ext/util.c b/tools/testing/selftests/sched_ext/util.c index 2111329ed289..c93fc5df552d 100644 --- a/tools/testing/selftests/sched_ext/util.c +++ b/tools/testing/selftests/sched_ext/util.c @@ -5,10 +5,21 @@ */ #include <errno.h> #include <fcntl.h> +#include <sched.h> +#include <signal.h> #include <stdio.h> #include <stdlib.h> #include <string.h> +#include <sys/prctl.h> +#include <sys/wait.h> #include <unistd.h> +#include <zlib.h> + +#include "util.h" + +#ifndef SCHED_EXT +#define SCHED_EXT 7 +#endif /* Returns read len on success, or -errno on failure. */ static ssize_t read_text(const char *path, char *buf, size_t max_len) @@ -69,3 +80,135 @@ int file_write_long(const char *path, long val) return 0; } + +struct scx_test_gated_worker scx_test_spawn_gated_worker(int cpu, bool set_sched_ext) +{ + struct scx_test_gated_worker worker = { .pid = -1, .start_fd = -1 }; + int ready[2], start[2]; + pid_t parent = getpid(); + char byte = 1; + + if (pipe(ready)) + return worker; + if (pipe(start)) { + close(ready[0]); + close(ready[1]); + return worker; + } + + worker.pid = fork(); + if (!worker.pid) { + struct sched_param param = {}; + cpu_set_t mask; + + close(ready[0]); + close(start[1]); + if (prctl(PR_SET_PDEATHSIG, SIGKILL) || getppid() != parent) + _exit(1); + CPU_ZERO(&mask); + CPU_SET(cpu, &mask); + if (sched_setaffinity(0, sizeof(mask), &mask)) + _exit(1); + if (set_sched_ext && sched_setscheduler(0, SCHED_EXT, ¶m)) + _exit(1); + if (write(ready[1], &byte, 1) != 1) + _exit(1); + close(ready[1]); + if (read(start[0], &byte, 1) != 1) + _exit(1); + close(start[0]); + for (;;) + asm volatile("" ::: "memory"); + } + if (worker.pid < 0) { + close(ready[0]); + close(ready[1]); + close(start[0]); + close(start[1]); + return worker; + } + + close(ready[1]); + close(start[0]); + if (read(ready[0], &byte, 1) != 1) { + close(ready[0]); + close(start[1]); + kill(worker.pid, SIGKILL); + waitpid(worker.pid, NULL, 0); + worker.pid = -1; + return worker; + } + close(ready[0]); + worker.start_fd = start[1]; + return worker; +} + +bool scx_test_start_gated_worker(struct scx_test_gated_worker *worker) +{ + char byte = 1; + + if (write(worker->start_fd, &byte, 1) != 1) + return false; + close(worker->start_fd); + worker->start_fd = -1; + return true; +} + +void scx_test_stop_gated_worker(struct scx_test_gated_worker *worker) +{ + if (worker->start_fd >= 0) + close(worker->start_fd); + if (worker->pid > 0) { + kill(worker->pid, SIGKILL); + waitpid(worker->pid, NULL, 0); + } + worker->pid = -1; + worker->start_fd = -1; +} + +static int config_preempt_lazy_mode(void) +{ + bool dynamic = false, lazy = false, immediate = false; + char buf[128]; + gzFile file; + + file = gzopen("/proc/config.gz", "r"); + if (!file) + return -1; + + while (gzgets(file, buf, sizeof(buf))) { + if (!strcmp(buf, "CONFIG_PREEMPT_DYNAMIC=y\n")) + dynamic = true; + else if (!strcmp(buf, "CONFIG_PREEMPT_LAZY=y\n")) + lazy = true; + else if (!strcmp(buf, "CONFIG_PREEMPT_NONE=y\n") || + !strcmp(buf, "CONFIG_PREEMPT_VOLUNTARY=y\n") || + !strcmp(buf, "CONFIG_PREEMPT=y\n") || + !strcmp(buf, "CONFIG_PREEMPT_RT=y\n")) + immediate = true; + } + gzclose(file); + + if (dynamic) + return -1; + if (lazy) + return 1; + if (immediate) + return 0; + return -1; +} + +int scx_test_preempt_lazy_mode(void) +{ + char buf[128]; + + if (read_text("/sys/kernel/debug/sched/preempt", buf, sizeof(buf)) > 0) { + if (strstr(buf, "(lazy)")) + return 1; + if (strstr(buf, "(none)") || strstr(buf, "(voluntary)") || + strstr(buf, "(full)")) + return 0; + } + + return config_preempt_lazy_mode(); +} diff --git a/tools/testing/selftests/sched_ext/util.h b/tools/testing/selftests/sched_ext/util.h index 681cec04b439..aac34fc4ef71 100644 --- a/tools/testing/selftests/sched_ext/util.h +++ b/tools/testing/selftests/sched_ext/util.h @@ -7,7 +7,19 @@ #ifndef __SCX_TEST_UTIL_H__ #define __SCX_TEST_UTIL_H__ +#include <stdbool.h> +#include <sys/types.h> + +struct scx_test_gated_worker { + pid_t pid; + int start_fd; +}; + long file_read_long(const char *path); int file_write_long(const char *path, long val); +struct scx_test_gated_worker scx_test_spawn_gated_worker(int cpu, bool set_sched_ext); +bool scx_test_start_gated_worker(struct scx_test_gated_worker *worker); +void scx_test_stop_gated_worker(struct scx_test_gated_worker *worker); +int scx_test_preempt_lazy_mode(void); #endif // __SCX_TEST_UTIL_H__ |
