diff options
Diffstat (limited to 'tools/testing/selftests')
| -rw-r--r-- | tools/testing/selftests/sched_ext/Makefile | 6 | ||||
| -rw-r--r-- | tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c | 60 | ||||
| -rw-r--r-- | tools/testing/selftests/sched_ext/cgroup_nr_cpus.c | 419 | ||||
| -rw-r--r-- | tools/testing/selftests/sched_ext/config | 1 |
4 files changed, 484 insertions, 2 deletions
diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile index ff0ccf71f30c..c5d3a2eaea7d 100644 --- a/tools/testing/selftests/sched_ext/Makefile +++ b/tools/testing/selftests/sched_ext/Makefile @@ -10,6 +10,7 @@ TEST_GEN_MODS_DIR := test_modules # override lib.mk's default rules OVERRIDE_TARGETS := 1 include ../lib.mk +include ../cgroup/lib/libcgroup.mk CURDIR := $(abspath .) REPOROOT := $(abspath ../../../..) @@ -154,7 +155,7 @@ $(INCLUDE_DIR)/%.bpf.skel.h: $(SCXOBJ_DIR)/%.bpf.o $(INCLUDE_DIR)/vmlinux.h $(BP override define CLEAN rm -rf $(OUTPUT_DIR) - rm -f $(TEST_GEN_PROGS) + rm -f $(TEST_GEN_PROGS) $(EXTRA_CLEAN) endef # Every testcase takes all of the BPF progs are dependencies by default. This @@ -163,6 +164,7 @@ endef all_test_bpfprogs := $(foreach prog,$(wildcard *.bpf.c),$(INCLUDE_DIR)/$(patsubst %.c,%.skel.h,$(prog))) auto-test-targets := \ + cgroup_nr_cpus \ create_dsq \ dequeue \ dequeue_iter \ @@ -217,7 +219,7 @@ $(testcase-targets): $(SCXOBJ_DIR)/%.o: %.c $(SCXOBJ_DIR)/runner.o $(all_test_bp $(SCXOBJ_DIR)/util.o: util.c | $(SCXOBJ_DIR) $(CC) $(CFLAGS) -c $< -o $@ -$(OUTPUT)/runner: $(SCXOBJ_DIR)/runner.o $(SCXOBJ_DIR)/util.o $(BPFOBJ) $(testcase-targets) +$(OUTPUT)/runner: $(SCXOBJ_DIR)/runner.o $(SCXOBJ_DIR)/util.o $(BPFOBJ) $(LIBCGROUP_O) $(testcase-targets) @echo "$(testcase-targets)" $(CC) $(CFLAGS) -o $@ $^ $(LDFLAGS) diff --git a/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c new file mode 100644 index 000000000000..841c83abb41b --- /dev/null +++ b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Validate scx_bpf_cgroup_nr_cpus() from both BPF_PROG_TYPE_SYSCALL and + * struct_ops contexts. + * + * Copyright (c) 2026 NVIDIA Corporation. + */ + +#include <scx/common.bpf.h> + +char _license[] SEC("license") = "GPL"; + +UEI_DEFINE(uei); + +/* input to cgroup_nr_cpus_read() */ +u64 query_cgid; +/* output of cgroup_nr_cpus_read(), -1 if @query_cgid couldn't be resolved */ +s64 query_nr_cpus = -1; + +/* recorded by ops.cgroup_init() for @init_cgid */ +u64 init_cgid; +s64 init_nr_cpus = -1; + +SEC("syscall") +int cgroup_nr_cpus_read(void *ctx) +{ + struct cgroup *cgrp; + + query_nr_cpus = -1; + + cgrp = bpf_cgroup_from_id(query_cgid); + if (!cgrp) + return -ENOENT; + + query_nr_cpus = scx_bpf_cgroup_nr_cpus(cgrp); + bpf_cgroup_release(cgrp); + + return 0; +} + +s32 BPF_STRUCT_OPS(cgroup_nr_cpus_cgroup_init, struct cgroup *cgrp, + struct scx_cgroup_init_args *args) +{ + if (cgrp->kn->id == init_cgid) + init_nr_cpus = scx_bpf_cgroup_nr_cpus(cgrp); + + return 0; +} + +void BPF_STRUCT_OPS(cgroup_nr_cpus_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SEC(".struct_ops.link") +struct sched_ext_ops cgroup_nr_cpus_ops = { + .cgroup_init = (void *)cgroup_nr_cpus_cgroup_init, + .exit = (void *)cgroup_nr_cpus_exit, + .name = "cgroup_nr_cpus", +}; diff --git a/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c new file mode 100644 index 000000000000..4973f7080353 --- /dev/null +++ b/tools/testing/selftests/sched_ext/cgroup_nr_cpus.c @@ -0,0 +1,419 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Verify that scx_bpf_cgroup_nr_cpus() reports the number of CPUs in a + * cgroup's effective cpuset, including inherited and updated cpusets. + * + * Copyright (c) 2026 NVIDIA Corporation. + */ + +#define _GNU_SOURCE +#include <bpf/bpf.h> +#include <errno.h> +#include <fcntl.h> +#include <limits.h> +#include <linux/limits.h> +#include <scx/common.h> +#include <stdbool.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> + +#include "cgroup_nr_cpus.bpf.skel.h" +#include "cgroup_util.h" +#include "scx_test.h" + +/* + * Hierarchy under the cgroup2 root, all with the cpu controller enabled so + * that ops.cgroup_init() runs for each of them: + * + * parent cpuset enabled by the root, enables cpuset for its children + * parent/child owns a cpuset + * parent/child/leaf no cpuset of its own, inherits child's + */ +struct cgroup_nr_cpus_ctx { + struct cgroup_nr_cpus *skel; + struct bpf_link *link; + char root[PATH_MAX]; + char parent[PATH_MAX]; + char child[PATH_MAX]; + char leaf[PATH_MAX]; + bool parent_created; + bool child_created; + bool leaf_created; +}; + +static int join_path(char *dst, size_t dst_size, const char *parent, const char *name) +{ + int ret; + + ret = snprintf(dst, dst_size, "%s/%s", parent, name); + if (ret < 0 || (size_t)ret >= dst_size) + return -ENAMETOOLONG; + return 0; +} + +static u64 cgroup_id(const char *path) +{ + union { + u64 id; + unsigned char bytes[8]; + } id = {}; + struct file_handle *handle; + int mount_id, ret; + + handle = calloc(1, sizeof(*handle) + sizeof(id)); + if (!handle) + return 0; + handle->handle_bytes = sizeof(id); + ret = name_to_handle_at(AT_FDCWD, path, handle, &mount_id, 0); + if (!ret && handle->handle_bytes == sizeof(id)) + memcpy(id.bytes, handle->f_handle, sizeof(id)); + free(handle); + + return ret ? 0 : id.id; +} + +/* + * Parse a cpulist such as "0-3,8,10-11". Return the number of CPUs and the + * lowest and highest CPU in @first and @last, or -errno on failure. + */ +static int parse_cpulist(const char *cpulist, u32 *first, u32 *last) +{ + const char *p = cpulist; + u32 lowest = UINT_MAX, highest = 0; + int count = 0; + + while (*p && *p != '\n') { + unsigned long start, end_cpu; + char *end; + + errno = 0; + start = strtoul(p, &end, 10); + if (errno || end == p || start > INT_MAX) + return -EINVAL; + end_cpu = start; + p = end; + if (*p == '-') { + end_cpu = strtoul(p + 1, &end, 10); + if (errno || end == p + 1 || end_cpu > INT_MAX || end_cpu < start) + return -EINVAL; + p = end; + } + if (end_cpu - start + 1 > (unsigned long)(INT_MAX - count)) + return -EOVERFLOW; + if (start < lowest) + lowest = start; + if (end_cpu > highest) + highest = end_cpu; + count += end_cpu - start + 1; + if (*p == ',') + p++; + else if (*p && *p != '\n') + return -EINVAL; + } + + if (first) + *first = lowest; + if (last) + *last = highest; + return count; +} + +/* + * Number of CPUs in @cgroup's cpuset.cpus.effective, or -errno. + * + * A sparse cpulist can exceed a page on large systems. cg_read() does a single + * bounded read, so size the buffer for the worst case of NR_CPUS=8192 and + * reject a read that fills it rather than parsing a truncated list. + */ +static int effective_nr_cpus(const char *cgroup, u32 *first, u32 *last) +{ + static char buf[65536]; + + if (cg_read(cgroup, "cpuset.cpus.effective", buf, sizeof(buf))) + return -EIO; + if (strlen(buf) >= sizeof(buf) - 1) + return -EOVERFLOW; + return parse_cpulist(buf, first, last); +} + +/* Run the SYSCALL program to sample scx_bpf_cgroup_nr_cpus() for @path. */ +static int kfunc_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, s64 *nr_cpus) +{ + LIBBPF_OPTS(bpf_test_run_opts, topts); + u64 cgid; + int err; + + cgid = cgroup_id(path); + if (!cgid) { + SCX_ERR("Failed to read cgroup ID of %s", path); + return -ENOENT; + } + + ctx->skel->bss->query_cgid = cgid; + err = bpf_prog_test_run_opts(bpf_program__fd(ctx->skel->progs.cgroup_nr_cpus_read), + &topts); + if (err || topts.retval) { + SCX_ERR("BPF_PROG_RUN failed for %s (err=%d retval=%d)", + path, err, (int)topts.retval); + return err ?: -EIO; + } + + *nr_cpus = ctx->skel->data->query_nr_cpus; + return 0; +} + +static bool check_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, int expected, + const char *what) +{ + s64 nr_cpus; + + if (kfunc_nr_cpus(ctx, path, &nr_cpus)) + return false; + if (nr_cpus != expected) { + SCX_ERR("%s: expected %d CPUs, got %lld", what, expected, + (long long)nr_cpus); + return false; + } + return true; +} + +/* + * Like check_nr_cpus() but tolerate a transient mismatch. A cpuset css being + * disabled stays attached to its cgroup until it's asynchronously offlined, and + * cpuset_num_cpus() keeps reporting its stale mask until then. + */ +static bool wait_nr_cpus(struct cgroup_nr_cpus_ctx *ctx, const char *path, int expected, + const char *what) +{ + s64 nr_cpus = -1; + int i; + + for (i = 0; i < 1000; i++) { + if (kfunc_nr_cpus(ctx, path, &nr_cpus)) + return false; + if (nr_cpus == expected) + return true; + usleep(1000); + } + SCX_ERR("%s: expected %d CPUs, got %lld", what, expected, (long long)nr_cpus); + return false; +} + +static bool controller_enabled(const char *cgroup, const char *file, const char *controller) +{ + char buf[4096], *saveptr, *token; + + if (cg_read(cgroup, file, buf, sizeof(buf))) + return false; + for (token = strtok_r(buf, "\n ", &saveptr); token; + token = strtok_r(NULL, "\n ", &saveptr)) + if (!strcmp(token, controller)) + return true; + return false; +} + +static void cleanup_ctx(struct cgroup_nr_cpus_ctx *ctx) +{ + bpf_link__destroy(ctx->link); + cgroup_nr_cpus__destroy(ctx->skel); + if (ctx->leaf_created) + cg_destroy(ctx->leaf); + if (ctx->child_created) + cg_destroy(ctx->child); + if (ctx->parent_created) + cg_destroy(ctx->parent); +} + +/* + * Enable @controller in the root's subtree_control if needed. Like the cgroup + * selftests, leave it enabled afterwards: the root is shared, and another + * manager may start relying on the controller while the test runs. + */ +static enum scx_test_status enable_controller(const char *root, const char *controller) +{ + char value[32]; + + if (controller_enabled(root, "cgroup.subtree_control", controller)) + return SCX_TEST_PASS; + if (!controller_enabled(root, "cgroup.controllers", controller)) + return SCX_TEST_SKIP; + + snprintf(value, sizeof(value), "+%s", controller); + if (cg_write(root, "cgroup.subtree_control", value)) + return SCX_TEST_SKIP; + return SCX_TEST_PASS; +} + +static enum scx_test_status setup_cgroups(struct cgroup_nr_cpus_ctx *ctx) +{ + enum scx_test_status status; + char name[64]; + + if (cg_find_unified_root(ctx->root, sizeof(ctx->root), NULL)) + return SCX_TEST_SKIP; + + status = enable_controller(ctx->root, "cpu"); + if (status != SCX_TEST_PASS) + return status; + status = enable_controller(ctx->root, "cpuset"); + if (status != SCX_TEST_PASS) + return status; + + snprintf(name, sizeof(name), "scx_nr_cpus_%d", getpid()); + if (join_path(ctx->parent, sizeof(ctx->parent), ctx->root, name) || + join_path(ctx->child, sizeof(ctx->child), ctx->parent, "child") || + join_path(ctx->leaf, sizeof(ctx->leaf), ctx->child, "leaf")) { + SCX_ERR("Cgroup path is too long"); + return SCX_TEST_FAIL; + } + + if (cg_create(ctx->parent)) { + SCX_ERR("Failed to create cgroup %s", ctx->parent); + return SCX_TEST_FAIL; + } + ctx->parent_created = true; + if (cg_write(ctx->parent, "cgroup.subtree_control", "+cpu +cpuset")) { + SCX_ERR("Failed to enable controllers in %s", ctx->parent); + return SCX_TEST_FAIL; + } + if (cg_create(ctx->child)) { + SCX_ERR("Failed to create cgroup %s", ctx->child); + return SCX_TEST_FAIL; + } + ctx->child_created = true; + if (cg_write(ctx->child, "cgroup.subtree_control", "+cpu")) { + SCX_ERR("Failed to enable cpu in %s", ctx->child); + return SCX_TEST_FAIL; + } + if (cg_create(ctx->leaf)) { + SCX_ERR("Failed to create cgroup %s", ctx->leaf); + return SCX_TEST_FAIL; + } + ctx->leaf_created = true; + + return SCX_TEST_PASS; +} + +static enum scx_test_status run(void *arg) +{ + struct cgroup_nr_cpus_ctx ctx = {}; + enum scx_test_status status; + char value[32]; + u32 first, last; + int nr_root, nr_child; + + (void)arg; + + /* + * SCX_ENUM_INIT() exits the process if vmlinux BTF can't be loaded, so + * run it before creating any cgroups that would then be left behind. + */ + ctx.skel = cgroup_nr_cpus__open(); + if (!ctx.skel) { + SCX_ERR("Failed to open skel"); + return SCX_TEST_FAIL; + } + SCX_ENUM_INIT(ctx.skel); + + status = setup_cgroups(&ctx); + if (status != SCX_TEST_PASS) + goto out; + status = SCX_TEST_FAIL; + + nr_root = effective_nr_cpus(ctx.root, NULL, NULL); + nr_child = effective_nr_cpus(ctx.child, &first, &last); + if (nr_root < 0 || nr_child < 0) { + SCX_ERR("Failed to read effective cpusets"); + goto out; + } + /* The effective cpuset can be empty, e.g. under a partition root. */ + if (nr_child < 2) { + status = SCX_TEST_SKIP; + goto out; + } + + ctx.skel->bss->init_cgid = cgroup_id(ctx.leaf); + if (!ctx.skel->bss->init_cgid) { + SCX_ERR("Failed to read cgroup ID of %s", ctx.leaf); + goto out; + } + if (cgroup_nr_cpus__load(ctx.skel)) { + SCX_ERR("Failed to load skel"); + goto out; + } + + /* The kfunc must be callable without a scheduler attached. */ + if (!check_nr_cpus(&ctx, ctx.root, nr_root, "root") || + !check_nr_cpus(&ctx, ctx.child, nr_child, "child") || + !check_nr_cpus(&ctx, ctx.leaf, nr_child, "inherited leaf")) + goto out; + + /* ops.cgroup_init() runs for existing cgroups when attaching. */ + ctx.link = bpf_map__attach_struct_ops(ctx.skel->maps.cgroup_nr_cpus_ops); + if (!ctx.link) { + SCX_ERR("Failed to attach scheduler"); + goto out; + } + if (ctx.skel->data->init_nr_cpus != nr_child) { + SCX_ERR("ops.cgroup_init(): expected %d CPUs, got %lld", nr_child, + (long long)ctx.skel->data->init_nr_cpus); + goto out; + } + + /* Non-contiguous cpuset, observed by the owner and by the inheritor. */ + if (nr_child > 2) { + snprintf(value, sizeof(value), "%u,%u", first, last); + if (cg_write(ctx.child, "cpuset.cpus", value)) { + SCX_ERR("Failed to set cpuset.cpus=%s for %s", value, ctx.child); + goto out; + } + if (!check_nr_cpus(&ctx, ctx.child, 2, "sparse child") || + !check_nr_cpus(&ctx, ctx.leaf, 2, "sparse inherited leaf")) + goto out; + } + + snprintf(value, sizeof(value), "%u", first); + if (cg_write(ctx.child, "cpuset.cpus", value)) { + SCX_ERR("Failed to set cpuset.cpus=%s for %s", value, ctx.child); + goto out; + } + if (!check_nr_cpus(&ctx, ctx.child, 1, "single-CPU child") || + !check_nr_cpus(&ctx, ctx.leaf, 1, "single-CPU inherited leaf")) + goto out; + + /* + * Disabling cpuset below @parent makes @child and @leaf inherit + * @parent's effective cpuset, which spans all of the root's CPUs. + */ + if (cg_write(ctx.parent, "cgroup.subtree_control", "-cpuset")) { + SCX_ERR("Failed to disable cpuset in %s", ctx.parent); + goto out; + } + nr_child = effective_nr_cpus(ctx.parent, NULL, NULL); + if (nr_child < 0) { + SCX_ERR("Failed to read effective cpuset of %s", ctx.parent); + goto out; + } + if (!wait_nr_cpus(&ctx, ctx.child, nr_child, "child after cpuset disable") || + !wait_nr_cpus(&ctx, ctx.leaf, nr_child, "leaf after cpuset disable")) + goto out; + + if (ctx.skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) { + SCX_ERR("Scheduler exited unexpectedly"); + goto out; + } + + status = SCX_TEST_PASS; +out: + cleanup_ctx(&ctx); + return status; +} + +struct scx_test cgroup_nr_cpus = { + .name = "cgroup_nr_cpus", + .description = "Verify scx_bpf_cgroup_nr_cpus() reports effective cpuset CPU counts", + .run = run, +}; +REGISTER_SCX_TEST(&cgroup_nr_cpus) diff --git a/tools/testing/selftests/sched_ext/config b/tools/testing/selftests/sched_ext/config index affa3cf33470..8173f9170ffe 100644 --- a/tools/testing/selftests/sched_ext/config +++ b/tools/testing/selftests/sched_ext/config @@ -1,6 +1,7 @@ CONFIG_SCHED_CLASS_EXT=y CONFIG_CGROUPS=y CONFIG_CGROUP_SCHED=y +CONFIG_CPUSETS=y CONFIG_EXT_GROUP_SCHED=y CONFIG_BPF=y CONFIG_BPF_SYSCALL=y |
