diff options
Diffstat (limited to 'tools')
66 files changed, 3474 insertions, 439 deletions
diff --git a/tools/arch/riscv/include/asm/csr.h b/tools/arch/riscv/include/asm/csr.h index 21d8cee04638..8df64314d613 100644 --- a/tools/arch/riscv/include/asm/csr.h +++ b/tools/arch/riscv/include/asm/csr.h @@ -163,12 +163,24 @@ #define HGATP_MODE_SHIFT HGATP32_MODE_SHIFT #endif -/* VSIP & HVIP relation */ +/* + * VSIP & HVIP relation + * + * The bit positions are same between VSIP and HVIP for interrupt + * numbers 13-63, where there's a shift for the SSI, STI and SEI. + */ #define VSIP_TO_HVIP_SHIFT (IRQ_VS_SOFT - IRQ_S_SOFT) -#define VSIP_VALID_MASK ((_AC(1, UL) << IRQ_S_SOFT) | \ +#define VSIP_BIAS_MASK ((_AC(1, UL) << IRQ_S_SOFT) | \ (_AC(1, UL) << IRQ_S_TIMER) | \ - (_AC(1, UL) << IRQ_S_EXT) | \ - (_AC(1, UL) << IRQ_PMU_OVF)) + (_AC(1, UL) << IRQ_S_EXT)) +#define VSIP_NO_BIAS_MASK (_AC(1, UL) << IRQ_PMU_OVF) +#define VSIP_VALID_MASK (VSIP_BIAS_MASK | VSIP_NO_BIAS_MASK) +#define vsip_to_hvip(_vsip) ((((_vsip) & VSIP_BIAS_MASK) << \ + VSIP_TO_HVIP_SHIFT) | \ + ((_vsip) & VSIP_NO_BIAS_MASK)) +#define hvip_to_vsip(_hvip) ((((_hvip) >> VSIP_TO_HVIP_SHIFT) & \ + VSIP_BIAS_MASK) | \ + ((_hvip) & VSIP_NO_BIAS_MASK)) /* AIA CSR bits */ #define TOPI_IID_SHIFT 16 diff --git a/tools/lib/bpf/libbpf.c b/tools/lib/bpf/libbpf.c index b749c01742ee..bfa64ae6c94d 100644 --- a/tools/lib/bpf/libbpf.c +++ b/tools/lib/bpf/libbpf.c @@ -6206,6 +6206,13 @@ bpf_object__relocate_core(struct bpf_object *obj, const char *targ_btf_path) return -EINVAL; insn = &prog->insns[insn_idx]; + if (is_ldimm64_insn(insn) && (size_t)insn_idx + 1 >= prog->insns_cnt) { + pr_warn("prog '%s': relo #%d: insn #%d (LDIMM64) is truncated\n", + prog->name, i, insn_idx); + err = -EINVAL; + goto out; + } + err = record_relo_core(prog, rec, insn_idx); if (err) { pr_warn("prog '%s': relo #%d: failed to record relocation: %s\n", diff --git a/tools/lib/bpf/relo_core.c b/tools/lib/bpf/relo_core.c index 8ad2715721cf..2672623a4198 100644 --- a/tools/lib/bpf/relo_core.c +++ b/tools/lib/bpf/relo_core.c @@ -980,23 +980,30 @@ done: } /* - * Turn instruction for which CO_RE relocation failed into invalid one with + * Turn instruction for which CO-RE relocation failed into invalid one with * distinct signature. */ -static void bpf_core_poison_insn(const char *prog_name, int relo_idx, - int insn_idx, struct bpf_insn *insn) +static int bpf_core_poison_insn(const char *prog_name, int relo_idx, + struct bpf_insn *insn, int insn_idx) { - pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n", - prog_name, relo_idx, insn_idx); - insn->code = BPF_JMP | BPF_CALL; - insn->dst_reg = 0; - insn->src_reg = 0; - insn->off = 0; - /* if this instruction is reachable (not a dead code), - * verifier will complain with the following message: - * invalid func unknown#195896080 - */ - insn->imm = 195896080; /* => 0xbad2310 => "bad relo" */ + int insn_cnt = is_ldimm64_insn(insn) ? 2 : 1; + int i; + + for (i = 0; i < insn_cnt; i++) { + pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n", + prog_name, relo_idx, insn_idx + i); + insn[i].code = BPF_JMP | BPF_CALL; + insn[i].dst_reg = 0; + insn[i].src_reg = 0; + insn[i].off = 0; + /* + * If this instruction is reachable (not dead code), the verifier + * will complain with "invalid func unknown#195896080". + */ + insn[i].imm = 195896080; /* => 0xbad2310 => "bad relo" */ + } + + return 0; } static int insn_bpf_size_to_bytes(struct bpf_insn *insn) @@ -1047,17 +1054,6 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, class = BPF_CLASS(insn->code); - if (res->poison) { -poison: - /* poison second part of ldimm64 to avoid confusing error from - * verifier about "unknown opcode 00" - */ - if (is_ldimm64_insn(insn)) - bpf_core_poison_insn(prog_name, relo_idx, insn_idx + 1, insn + 1); - bpf_core_poison_insn(prog_name, relo_idx, insn_idx, insn); - return 0; - } - orig_val = res->orig_val; new_val = res->new_val; @@ -1065,7 +1061,9 @@ poison: case BPF_ALU: case BPF_ALU64: if (BPF_SRC(insn->code) != BPF_K) - return -EINVAL; + goto bad_insn; + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); if (res->validate && insn->imm != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (ALU/ALU64) value: got %d, exp %llu -> %llu\n", prog_name, relo_idx, @@ -1082,6 +1080,8 @@ poison: case BPF_LDX: case BPF_ST: case BPF_STX: + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); if (res->validate && insn->off != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDX/ST/STX) value: got %d, exp %llu -> %llu\n", prog_name, relo_idx, insn_idx, insn->off, (unsigned long long)orig_val, @@ -1097,7 +1097,7 @@ poison: pr_warn("prog '%s': relo #%d: insn #%d (LDX/ST/STX) accesses field incorrectly. " "Make sure you are accessing pointers, unsigned integers, or fields of matching type and size.\n", prog_name, relo_idx, insn_idx); - goto poison; + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); } orig_val = insn->off; @@ -1140,6 +1140,9 @@ poison: return -EINVAL; } + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); + imm = (__u32)insn[0].imm | ((__u64)insn[1].imm << 32); if (res->validate && imm != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDIMM64) value: got %llu, exp %llu -> %llu\n", @@ -1157,6 +1160,7 @@ poison: break; } default: +bad_insn: pr_warn("prog '%s': relo #%d: trying to relocate unrecognized insn #%d, code:0x%x, src:0x%x, dst:0x%x, off:0x%x, imm:0x%x\n", prog_name, relo_idx, insn_idx, insn->code, (unsigned)insn->src_reg, (unsigned)insn->dst_reg, (unsigned)insn->off, (unsigned)insn->imm); diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 2ddb01a059fd..22f24ebef8a9 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -106,7 +106,7 @@ u64 scx_bpf_now(void) __ksym __weak; void scx_bpf_events(struct scx_event_stats *events, size_t events__sz) __ksym __weak; s32 scx_bpf_cpu_to_cid(s32 cpu) __ksym __weak; s32 scx_bpf_cid_to_cpu(s32 cid) __ksym __weak; -void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out) __ksym __weak; +void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out, size_t out__sz) __ksym __weak; void scx_bpf_kick_cid(s32 cid, u64 flags) __ksym __weak; s32 scx_bpf_task_cid(const struct task_struct *p) __ksym __weak; s32 scx_bpf_this_cid(void) __ksym __weak; diff --git a/tools/sched_ext/scx_qmap.bpf.c b/tools/sched_ext/scx_qmap.bpf.c index 67b7c01cae55..2f3653199842 100644 --- a/tools/sched_ext/scx_qmap.bpf.c +++ b/tools/sched_ext/scx_qmap.bpf.c @@ -961,9 +961,6 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init_task, struct task_struct *p, taskc->highpri = false; taskc->core_sched_seq = 0; cmask_init(&taskc->cpus_allowed, 0, scx_bpf_nr_cids()); - bpf_rcu_read_lock(); - cmask_from_cpumask(&taskc->cpus_allowed, p->cpus_ptr); - bpf_rcu_read_unlock(); v = bpf_task_storage_get(&task_ctx_stor, p, NULL, BPF_LOCAL_STORAGE_GET_F_CREATE); diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile index 2d960626750e..273853937c25 100644 --- a/tools/testing/selftests/Makefile +++ b/tools/testing/selftests/Makefile @@ -35,8 +35,11 @@ TARGETS += fchmodat2 TARGETS += filesystems TARGETS += filesystems/binderfs TARGETS += filesystems/epoll +TARGETS += filesystems/eventfd TARGETS += filesystems/failfs TARGETS += filesystems/fat +TARGETS += filesystems/openat2 +TARGETS += filesystems/open_tree_ns TARGETS += filesystems/overlayfs TARGETS += filesystems/statmount TARGETS += filesystems/mount-notify @@ -46,7 +49,7 @@ TARGETS += filesystems/move_mount TARGETS += filesystems/empty_mntns TARGETS += filesystems/fsmount_ns TARGETS += filesystems/fscontext_ns -TARGETS += filesystems/mntns_cleanup +TARGETS += filesystems/xattr TARGETS += firmware TARGETS += fpu TARGETS += ftrace @@ -103,7 +106,6 @@ TARGETS += prctl TARGETS += proc TARGETS += pstore TARGETS += ptrace -TARGETS += openat2 TARGETS += rdma TARGETS += resctrl TARGETS += riscv diff --git a/tools/testing/selftests/bpf/prog_tests/cb_refs.c b/tools/testing/selftests/bpf/prog_tests/cb_refs.c index 78566b817fd7..490e15e7126d 100644 --- a/tools/testing/selftests/bpf/prog_tests/cb_refs.c +++ b/tools/testing/selftests/bpf/prog_tests/cb_refs.c @@ -13,7 +13,7 @@ struct { } cb_refs_tests[] = { { "underflow_prog", "release kfunc bpf_kfunc_call_test_release expects referenced PTR_TO_BTF_ID passed to R1" }, { "leak_prog", "Possibly NULL pointer passed to helper R2" }, - { "nested_cb", "Unreleased reference id=4 alloc_insn=2" }, /* alloc_insn=2{4,5} */ + { "nested_cb", "Unreleased reference id=5 alloc_insn=2" }, /* alloc_insn=2{4,5} */ { "non_cb_transfer_ref", "Unreleased reference id=4 alloc_insn=1" }, /* alloc_insn=1{1,2} */ }; diff --git a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c index a18d3680fb16..51f42b02a267 100644 --- a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c +++ b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c @@ -14,6 +14,197 @@ static char log[16 * 1024]; +static int load_core_relo_insns(int btf_fd, struct bpf_insn *insns, int insn_cnt, + struct bpf_func_info *funcs, int func_cnt, + int enum_id, int access_str_off, int insn_idx, + bool relocate) +{ + struct bpf_core_relo relo = { + .insn_off = insn_idx * sizeof(struct bpf_insn), + .type_id = enum_id, + .access_str_off = access_str_off, + .kind = BPF_CORE_ENUMVAL_VALUE, + }; + union bpf_attr attr = { + .prog_type = BPF_PROG_TYPE_SOCKET_FILTER, + .insn_cnt = insn_cnt, + .insns = (__u64)insns, + .license = (__u64)"GPL", + .log_buf = (__u64)log, + .log_size = sizeof(log), + .log_level = 2, + .prog_btf_fd = btf_fd, + .func_info_rec_size = sizeof(struct bpf_func_info), + .func_info = (__u64)funcs, + .func_info_cnt = func_cnt, + }; + + if (relocate) { + attr.core_relo_cnt = 1; + attr.core_relos = (__u64)&relo; + attr.core_relo_rec_size = sizeof(relo); + } + memset(log, 0, sizeof(log)); + return sys_bpf_prog_load(&attr, sizeof(attr), 1); +} + +static void test_early_core_relo(void) +{ + static const char unrecognized[] = "trying to relocate unrecognized insn #2"; + static const struct { + const char *name; + struct bpf_insn insns[2]; + const char *err_msg; + } tests[] = { + { "poison_exit", { BPF_EXIT_INSN() }, unrecognized }, + { "poison_ja", { BPF_JMP_A(1) }, unrecognized }, + { "poison_jmp", { BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized }, + { "poison_jmp32", { BPF_JMP32_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized }, + { "poison_call", { BPF_EMIT_CALL(BPF_FUNC_get_prandom_u32) }, unrecognized }, + { "poison_alu_reg", { BPF_MOV32_REG(BPF_REG_0, BPF_REG_1) }, unrecognized }, + { "poison_alu64_reg", { BPF_MOV64_REG(BPF_REG_0, BPF_REG_1) }, unrecognized }, + { "poison_ld_abs", { BPF_LD_ABS(BPF_W, 0) }, + "insn #2 (LDIMM64) has unexpected form" }, + { "poison_alu_imm", { BPF_MOV32_IMM(BPF_REG_0, 0) } }, + { "poison_alu64_imm", { BPF_MOV64_IMM(BPF_REG_0, 0) } }, + { "poison_ldx", { BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0) } }, + { "poison_st", { BPF_ST_MEM(BPF_W, BPF_REG_10, -4, 0) } }, + { "poison_stx", { BPF_STX_MEM(BPF_W, BPF_REG_10, BPF_REG_0, -4) } }, + { "poison_ldimm64", { BPF_LD_IMM64(BPF_REG_0, 0) } }, + }; + struct test_btf { + struct btf_header hdr; + __u32 types[18]; + char strings[64]; + } raw_btf = { + .hdr = { + .magic = BTF_MAGIC, + .version = BTF_VERSION, + .hdr_len = sizeof(struct btf_header), + .type_off = 0, + .type_len = sizeof(raw_btf.types), + .str_off = offsetof(struct test_btf, strings) - + offsetof(struct test_btf, types), + .str_len = sizeof(raw_btf.strings), + }, + .types = { + BTF_TYPE_INT_ENC(1, BTF_INT_SIGNED, 0, 32, 4), /* [1] int */ + BTF_FUNC_PROTO_ENC(1, 0), /* [2] int (*)(void) */ + BTF_FUNC_ENC(5, 2), /* [3] main_fn */ + BTF_FUNC_ENC(13, 2), /* [4] sub_fn */ + BTF_TYPE_ENC(20, BTF_INFO_ENC(BTF_KIND_ENUM, 0, 1), 4), /* [5] enum */ + BTF_ENUM_ENC(45, 0), /* value = 0 */ + }, + .strings = "\0int\0main_fn\0sub_fn\0core_relo_poison_missing\0value\0" "0", + }; + struct bpf_func_info funcs[] = { + { .insn_off = 0, .type_id = 3 }, + { .insn_off = 3, .type_id = 4 }, + }; + struct bpf_insn core_only[] = { + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + struct bpf_insn subprog[] = { + BPF_CALL_REL(2), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + struct bpf_insn truncated_ldimm64[] = { + BPF_RAW_INSN(BPF_LD | BPF_IMM | BPF_DW, 0, 0, 0, 0), + }; + int access_str_off = 51; /* offset of "0" */ + int enum_id = 5; + int btf_fd, prog_fd = -1, i; + + btf_fd = bpf_btf_load(&raw_btf, sizeof(raw_btf), NULL); + if (!ASSERT_GE(btf_fd, 0, "btf_load")) + goto cleanup; + + if (test__start_subtest("without_func_info")) { + prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0, + enum_id, access_str_off, 2, false); + if (!ASSERT_GE(prog_fd, 0, "control_load")) + goto cleanup; + close(prog_fd); + prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0, + enum_id, access_str_off, 2, true); + if (!ASSERT_GE(prog_fd, 0, "poisoned_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + close(prog_fd); + prog_fd = -1; + } + + if (test__start_subtest("before_subprog_validation")) { + prog_fd = load_core_relo_insns(btf_fd, subprog, ARRAY_SIZE(subprog), funcs, 2, + enum_id, access_str_off, 2, true); + if (!ASSERT_LT(prog_fd, 0, "poisoned_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + ASSERT_HAS_SUBSTR(log, "last insn is not an exit or jmp", "poisoned_load_log"); + } + + if (test__start_subtest("truncated_ldimm64")) { + prog_fd = load_core_relo_insns(btf_fd, truncated_ldimm64, + ARRAY_SIZE(truncated_ldimm64), NULL, 0, + enum_id, access_str_off, 0, true); + if (!ASSERT_LT(prog_fd, 0, "truncated_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "invalid bpf_ld_imm64 insn", "truncated_load_log"); + } + + for (i = 0; i < ARRAY_SIZE(tests); i++) { + struct bpf_insn insns[] = { + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1), + tests[i].insns[0], + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + bool is_ldimm64 = insns[2].code == (BPF_LD | BPF_DW | BPF_IMM); + + if (!test__start_subtest(tests[i].name)) + continue; + if (is_ldimm64) { + insns[1].off = 2; + insns[3] = tests[i].insns[1]; + } + prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1, + enum_id, access_str_off, 2, false); + if (!ASSERT_GE(prog_fd, 0, "control_load")) + goto cleanup; + close(prog_fd); + prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1, + enum_id, access_str_off, 2, true); + if (!tests[i].err_msg) { + ASSERT_GE(prog_fd, 0, "dead_poison_load"); + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + if (is_ldimm64) + ASSERT_HAS_SUBSTR(log, "substituting insn #3", "poison_ldimm64_log"); + } else { + ASSERT_LT(prog_fd, 0, "invalid_poison_load"); + ASSERT_HAS_SUBSTR(log, tests[i].err_msg, "invalid_poison_log"); + ASSERT_NULL(strstr(log, "substituting insn"), "invalid_poison_substitution"); + } + close(prog_fd); + prog_fd = -1; + } + +cleanup: + if (env.verbosity > VERBOSE_NORMAL && log[0]) { + printf("-------- program load log start --------\n"); + printf("%s", log); + printf("-------- program load log end ----------\n"); + } + close(prog_fd); + close(btf_fd); +} + /* Check that verifier rejects BPF program containing relocation * pointing to non-existent BTF type. */ @@ -120,6 +311,7 @@ out: void test_core_reloc_raw(void) { + test_early_core_relo(); if (test__start_subtest("bad_local_id")) test_bad_local_id(); } diff --git a/tools/testing/selftests/bpf/prog_tests/dynptr.c b/tools/testing/selftests/bpf/prog_tests/dynptr.c index 5fda11590708..4396560365e8 100644 --- a/tools/testing/selftests/bpf/prog_tests/dynptr.c +++ b/tools/testing/selftests/bpf/prog_tests/dynptr.c @@ -9,6 +9,7 @@ enum test_setup_type { SETUP_SYSCALL_SLEEP, SETUP_SKB_PROG, + SETUP_SKB_PROG_NONLINEAR, SETUP_SKB_PROG_TP, SETUP_XDP_PROG, }; @@ -32,6 +33,7 @@ static struct { {"test_ringbuf", SETUP_SYSCALL_SLEEP}, {"test_skb_readonly", SETUP_SKB_PROG}, {"test_dynptr_skb_data", SETUP_SKB_PROG}, + {"test_dynptr_skb_slice_non_linear", SETUP_SKB_PROG_NONLINEAR}, {"test_dynptr_skb_meta_data", SETUP_SKB_PROG}, {"test_dynptr_skb_meta_flags", SETUP_SKB_PROG}, {"test_adjust", SETUP_SYSCALL_SLEEP}, @@ -94,7 +96,9 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ bpf_link__destroy(link); break; case SETUP_SKB_PROG: + case SETUP_SKB_PROG_NONLINEAR: { + struct __sk_buff ctx = {}; int prog_fd; char buf[64]; @@ -106,6 +110,12 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ .repeat = 1, ); + if (setup_type == SETUP_SKB_PROG_NONLINEAR) { + ctx.data_end = ETH_HLEN + sizeof(struct iphdr); + topts.ctx_in = &ctx; + topts.ctx_size_in = sizeof(ctx); + } + prog_fd = bpf_program__fd(prog); if (!ASSERT_GE(prog_fd, 0, "prog_fd")) goto cleanup; diff --git a/tools/testing/selftests/bpf/prog_tests/exceptions.c b/tools/testing/selftests/bpf/prog_tests/exceptions.c index 3588d6f97fd4..639866ce09a9 100644 --- a/tools/testing/selftests/bpf/prog_tests/exceptions.c +++ b/tools/testing/selftests/bpf/prog_tests/exceptions.c @@ -55,6 +55,7 @@ static void test_exceptions_success(void) RUN_SUCCESS(exception_ext, 0); RUN_SUCCESS(exception_ext_mod_cb_runtime, 35); RUN_SUCCESS(exception_throw_subprog, 1); + RUN_SUCCESS(exception_throw_subprog_stack_cb, 0x1234); RUN_SUCCESS(exception_assert_nz_gfunc, 1); RUN_SUCCESS(exception_assert_zero_gfunc, 1); RUN_SUCCESS(exception_assert_neg_gfunc, 1); diff --git a/tools/testing/selftests/bpf/prog_tests/linked_list.c b/tools/testing/selftests/bpf/prog_tests/linked_list.c index c3d133c6a00d..52fabbee3dd5 100644 --- a/tools/testing/selftests/bpf/prog_tests/linked_list.c +++ b/tools/testing/selftests/bpf/prog_tests/linked_list.c @@ -714,7 +714,7 @@ static void test_btf(void) break; err = btf__load_into_kernel(btf); - ASSERT_EQ(err, -ELOOP, "check btf"); + ASSERT_EQ(err, 0, "check btf"); btf__free(btf); break; } @@ -773,7 +773,7 @@ static void test_btf(void) break; err = btf__load_into_kernel(btf); - ASSERT_EQ(err, -ELOOP, "check btf"); + ASSERT_EQ(err, 0, "check btf"); btf__free(btf); break; } diff --git a/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c new file mode 100644 index 000000000000..a487aa68f2ee --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c @@ -0,0 +1,296 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */ + +#include <bpf/btf.h> +#include <linux/btf.h> +#include <test_progs.h> + +#define SPIN_LOCK 2 +#define LIST_HEAD 3 +#define LIST_NODE 4 +/* Keep in sync with BTF_MAX_OWNERSHIP_DEPTH. */ +#define MAX_OWNERSHIP_DEPTH 8 + +static struct btf *init_btf(void) +{ + struct btf *btf; + int id; + + btf = btf__new_empty(); + if (!ASSERT_OK_PTR(btf, "btf__new_empty")) + return NULL; + id = btf__add_int(btf, "int", 4, BTF_INT_SIGNED); + if (!ASSERT_EQ(id, 1, "btf__add_int")) + goto err_out; + id = btf__add_struct(btf, "bpf_spin_lock", 4); + if (!ASSERT_EQ(id, SPIN_LOCK, "btf__add_struct bpf_spin_lock")) + goto err_out; + id = btf__add_struct(btf, "bpf_list_head", 16); + if (!ASSERT_EQ(id, LIST_HEAD, "btf__add_struct bpf_list_head")) + goto err_out; + id = btf__add_struct(btf, "bpf_list_node", 24); + if (!ASSERT_EQ(id, LIST_NODE, "btf__add_struct bpf_list_node")) + goto err_out; + return btf; + +err_out: + btf__free(btf); + return NULL; +} + +static int add_local_kptr(struct btf *btf, int pointee_id, const char *tag) +{ + int id; + + id = btf__add_type_tag(btf, tag, pointee_id); + if (!ASSERT_GT(id, 0, "btf__add_type_tag")) + return id; + id = btf__add_ptr(btf, id); + ASSERT_GT(id, 0, "btf__add_ptr"); + return id; +} + +static void test_self_cycle(const char *tag, int expected_err) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 7, tag); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "self_cycle", 8); + if (!ASSERT_EQ(id, 7, "btf__add_struct self_cycle")) + goto out; + err = btf__add_field(btf, "next", 6, 0, 0); + if (!ASSERT_OK(err, "btf__add_field self_cycle::next")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +static void test_aba_cycle(void) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 10, "kptr"); + if (id <= 0) + goto out; + id = add_local_kptr(btf, 9, "kptr"); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "cycle_a", 8); + if (!ASSERT_EQ(id, 9, "btf__add_struct cycle_a")) + goto out; + err = btf__add_field(btf, "b", 6, 0, 0); + if (!ASSERT_OK(err, "btf__add_field cycle_a::b")) + goto out; + id = btf__add_struct(btf, "cycle_b", 8); + if (!ASSERT_EQ(id, 10, "btf__add_struct cycle_b")) + goto out; + err = btf__add_field(btf, "a", 8, 0, 0); + if (!ASSERT_OK(err, "btf__add_field cycle_b::a")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, -ELOOP, "check btf"); +out: + btf__free(btf); +} + +static void test_mixed_cycle(void) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 7, "kptr"); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "mixed_owner", 20); + if (!ASSERT_EQ(id, 7, "btf__add_struct mixed_owner")) + goto out; + err = btf__add_field(btf, "root", LIST_HEAD, 0, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_owner::root")) + goto out; + err = btf__add_field(btf, "lock", SPIN_LOCK, 128, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_owner::lock")) + goto out; + id = btf__add_decl_tag(btf, "contains:mixed_node:node", 7, 0); + if (!ASSERT_EQ(id, 8, "btf__add_decl_tag mixed_owner")) + goto out; + id = btf__add_struct(btf, "mixed_node", 32); + if (!ASSERT_EQ(id, 9, "btf__add_struct mixed_node")) + goto out; + err = btf__add_field(btf, "node", LIST_NODE, 0, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_node::node")) + goto out; + err = btf__add_field(btf, "owner", 6, 192, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_node::owner")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, -ELOOP, "check btf"); +out: + btf__free(btf); +} + +static void test_acyclic_depth(int depth, bool child_first, bool shared_suffix, int expected_err) +{ + int ptr_id[MAX_OWNERSHIP_DEPTH + 1]; + int first_struct_id; + struct btf *btf; + int id, err, i, n, pointee_id; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + first_struct_id = 5 + 2 * depth; + for (i = 0; i < depth; i++) { + if (i == depth - 1) + pointee_id = first_struct_id + depth; + else + pointee_id = first_struct_id + (child_first ? depth - 2 - i : i + 1); + ptr_id[i] = add_local_kptr(btf, pointee_id, "kptr"); + if (ptr_id[i] <= 0) + goto out; + } + for (n = 0; n < depth; n++) { + char name[32]; + int offset = 0; + + i = child_first ? depth - 1 - n : n; + snprintf(name, sizeof(name), "owner_%d", i); + id = btf__add_struct(btf, name, shared_suffix && !i ? 16 : 8); + if (!ASSERT_EQ(id, first_struct_id + n, "btf__add_struct owner")) + goto out; + if (shared_suffix && !i) { + /* + * Visit the shared suffix through the shorter path before + * reaching it again with less remaining depth. + */ + err = btf__add_field(btf, "suffix", ptr_id[1], 0, 0); + if (!ASSERT_OK(err, "btf__add_field owner::suffix")) + goto out; + offset = 64; + } + err = btf__add_field(btf, "next", ptr_id[i], offset, 0); + if (!ASSERT_OK(err, "btf__add_field owner::next")) + goto out; + } + id = btf__add_struct(btf, "plain_leaf", 4); + if (!ASSERT_EQ(id, first_struct_id + depth, "btf__add_struct plain_leaf")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +static void test_graph_depth(bool rbtree, int depth, int expected_err) +{ + int root_type = LIST_HEAD, node_type = LIST_NODE, node_size = 24; + int id, err, i, lock_off, root_off, size; + struct btf *btf; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + if (rbtree) { + root_type = btf__add_struct(btf, "bpf_rb_root", 16); + if (!ASSERT_GT(root_type, 0, "btf__add_struct bpf_rb_root")) + goto out; + node_type = btf__add_struct(btf, "bpf_rb_node", 32); + if (!ASSERT_GT(node_type, 0, "btf__add_struct bpf_rb_node")) + goto out; + node_size = 32; + } + + for (i = 0; i < depth; i++) { + char name[32], tag[64]; + + lock_off = i ? node_size : 0; + root_off = lock_off + 8; + size = i == depth - 1 ? node_size : root_off + 16; + snprintf(name, sizeof(name), "graph_owner_%d", i); + id = btf__add_struct(btf, name, size); + if (!ASSERT_GT(id, 0, "btf__add_struct graph_owner")) + goto out; + if (i) { + err = btf__add_field(btf, "node", node_type, 0, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::node")) + goto out; + } + if (i == depth - 1) + continue; + err = btf__add_field(btf, "lock", SPIN_LOCK, lock_off * 8, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::lock")) + goto out; + err = btf__add_field(btf, "root", root_type, root_off * 8, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::root")) + goto out; + snprintf(tag, sizeof(tag), "contains:graph_owner_%d:node", i + 1); + err = btf__add_decl_tag(btf, tag, id, i ? 2 : 1); + if (!ASSERT_GT(err, 0, "btf__add_decl_tag graph_owner")) + goto out; + } + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +void test_local_kptr_ownership(void) +{ + if (test__start_subtest("self_cycle")) + test_self_cycle("kptr", -ELOOP); + if (test__start_subtest("untrusted_self_cycle")) + test_self_cycle("kptr_untrusted", 0); + if (test__start_subtest("percpu_self_cycle")) + test_self_cycle("percpu_kptr", -ELOOP); + if (test__start_subtest("ABA_cycle")) + test_aba_cycle(); + if (test__start_subtest("mixed_graph_root_cycle")) + test_mixed_cycle(); + if (test__start_subtest("max_acyclic")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, false, 0); + if (test__start_subtest("too_deep_acyclic")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, false, -ELOOP); + if (test__start_subtest("max_acyclic_child_first")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, true, false, 0); + if (test__start_subtest("too_deep_acyclic_child_first")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, true, false, -ELOOP); + if (test__start_subtest("max_acyclic_shared_suffix")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, true, 0); + if (test__start_subtest("too_deep_acyclic_shared_suffix")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, true, -ELOOP); + if (test__start_subtest("list_three_types")) + test_graph_depth(false, 3, 0); + if (test__start_subtest("list_four_types")) + test_graph_depth(false, 4, 0); + if (test__start_subtest("list_max_depth")) + test_graph_depth(false, MAX_OWNERSHIP_DEPTH, 0); + if (test__start_subtest("list_too_deep")) + test_graph_depth(false, MAX_OWNERSHIP_DEPTH + 1, -ELOOP); + if (test__start_subtest("rbtree_three_types")) + test_graph_depth(true, 3, 0); + if (test__start_subtest("rbtree_four_types")) + test_graph_depth(true, 4, 0); + if (test__start_subtest("rbtree_max_depth")) + test_graph_depth(true, MAX_OWNERSHIP_DEPTH, 0); + if (test__start_subtest("rbtree_too_deep")) + test_graph_depth(true, MAX_OWNERSHIP_DEPTH + 1, -ELOOP); +} diff --git a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c index a72ae0b29f6e..7b4a1e24363b 100644 --- a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c +++ b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c @@ -1,4 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 +#define _GNU_SOURCE +#include <sched.h> #include <test_progs.h> #include "cgroup_helpers.h" #include "percpu_alloc_array.skel.h" @@ -350,6 +352,103 @@ static void test_lru_percpu_hash_cpu_flag(void) test_percpu_map_cpu_flag(BPF_MAP_TYPE_LRU_PERCPU_HASH); } +/* + * A BPF_F_CPU update that creates an element must zero the value on the other + * cpus, rather than leave them holding whatever the recycled element last + * contained. max_entries is 1 so the second key can only reuse the element + * the first one released. + */ +static void test_percpu_map_cpu_flag_create(enum bpf_map_type map_type, __u32 map_flags) +{ + LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = map_flags); + const u32 stale = 0xDEADC0DE, fresh = 0xC0FFEE; + int nr_cpus, cpu, map_fd, err, key; + int pinned_cpu, value_cpu; + cpu_set_t old_mask, new_mask; + bool restore_mask = false; + u32 value; + u64 flags; + + nr_cpus = libbpf_num_possible_cpus(); + if (!ASSERT_GT(nr_cpus, 0, "libbpf_num_possible_cpus")) + return; + + if (nr_cpus < 2) { + test__skip(); + return; + } + + map_fd = bpf_map_create(map_type, "cpu_flag_create", sizeof(key), sizeof(value), 1, &opts); + if (!ASSERT_GE(map_fd, 0, "bpf_map_create")) + return; + + /* NO_PREALLOC recycles per cpu, so keep the delete and the create on one cpu. */ + err = sched_getaffinity(0, sizeof(old_mask), &old_mask); + if (!ASSERT_OK(err, "sched_getaffinity")) + goto out; + + pinned_cpu = sched_getcpu(); + if (!ASSERT_GE(pinned_cpu, 0, "sched_getcpu")) + goto out; + + CPU_ZERO(&new_mask); + CPU_SET(pinned_cpu, &new_mask); + err = sched_setaffinity(0, sizeof(new_mask), &new_mask); + if (!ASSERT_OK(err, "sched_setaffinity")) + goto out; + restore_mask = true; + + value_cpu = pinned_cpu ? 0 : 1; + + key = 1; + value = stale; + err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS); + if (!ASSERT_OK(err, "bpf_map_update_elem all_cpus")) + goto out; + + err = bpf_map_delete_elem(map_fd, &key); + if (!ASSERT_OK(err, "bpf_map_delete_elem")) + goto out; + + key = 2; + value = fresh; + flags = (u64)value_cpu << 32 | BPF_F_CPU; + err = bpf_map_update_elem(map_fd, &key, &value, flags); + if (!ASSERT_OK(err, "bpf_map_update_elem specified cpu")) + goto out; + + for (cpu = 0; cpu < nr_cpus; cpu++) { + value = 0; + flags = (u64)cpu << 32 | BPF_F_CPU; + err = bpf_map_lookup_elem_flags(map_fd, &key, &value, flags); + if (!ASSERT_OK(err, "bpf_map_lookup_elem_flags specified cpu")) + goto out; + if (!ASSERT_EQ(value, cpu == value_cpu ? fresh : 0, "value on specified cpu")) + goto out; + } + +out: + if (restore_mask) + sched_setaffinity(0, sizeof(old_mask), &old_mask); + close(map_fd); +} + +static void test_percpu_hash_cpu_flag_create(void) +{ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, 0); +} + +static void test_percpu_hash_cpu_flag_create_malloc(void) +{ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, BPF_F_NO_PREALLOC); +} + +static void test_lru_percpu_hash_cpu_flag_create(void) +{ + /* lru without prealloc is -ENOTSUPP, so there is no malloc variant */ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_LRU_PERCPU_HASH, 0); +} + static void test_percpu_cgroup_storage_cpu_flag(void) { struct percpu_alloc_array *skel = NULL; @@ -454,6 +553,12 @@ void test_percpu_alloc(void) test_percpu_hash_cpu_flag(); if (test__start_subtest("cpu_flag_lru_percpu_hash")) test_lru_percpu_hash_cpu_flag(); + if (test__start_subtest("cpu_flag_create_percpu_hash")) + test_percpu_hash_cpu_flag_create(); + if (test__start_subtest("cpu_flag_create_percpu_hash_malloc")) + test_percpu_hash_cpu_flag_create_malloc(); + if (test__start_subtest("cpu_flag_create_lru_percpu_hash")) + test_lru_percpu_hash_cpu_flag_create(); if (test__start_subtest("cpu_flag_percpu_cgroup_storage")) test_percpu_cgroup_storage_cpu_flag(); if (test__start_subtest("cpu_flag_array")) diff --git a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c index 9c11938fe597..78d642a02bdb 100644 --- a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c +++ b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c @@ -1,4 +1,5 @@ // SPDX-License-Identifier: GPL-2.0 +#include <poll.h> #include <test_progs.h> #include <bpf/bpf_endian.h> @@ -110,6 +111,122 @@ cleanup: close(serv); } +static void test_tcp_listen_pending(struct sock_destroy_prog *skel) +{ + int serv = -1, clien = -1, accept_serv = -1, n, serv_port; + struct pollfd pfd = { .events = POLLIN }; + char buf[1]; + + serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0); + if (!ASSERT_GE(serv, 0, "start_server")) + goto cleanup; + serv_port = get_socket_local_port(serv); + if (!ASSERT_GE(serv_port, 0, "get_sock_local_port")) + goto cleanup; + skel->bss->serv_port = (__be16)serv_port; + + /* + * Connect but never accept, so the child sits in the accept queue + * of the listener. Wait until it's actually there. + */ + clien = connect_to_fd(serv, 0); + if (!ASSERT_GE(clien, 0, "connect_to_fd")) + goto cleanup; + pfd.fd = serv; + if (!ASSERT_EQ(poll(&pfd, 1, -1), 1, "poll listener")) + goto cleanup; + + /* Run iterator program that destroys server sockets. */ + start_iter_sockets(skel->progs.iter_tcp6_server); + + accept_serv = accept(serv, NULL, NULL); + if (!ASSERT_LT(accept_serv, 0, "accept on destroyed listener")) + goto cleanup; + ASSERT_EQ(errno, EINVAL, "error code on destroyed listener"); + + /* The unaccepted child was reset along with the listener. */ + n = recv(clien, buf, sizeof(buf), 0); + if (!ASSERT_LT(n, 0, "client recv on reset child")) + goto cleanup; + ASSERT_EQ(errno, ECONNRESET, "error code on reset child"); + +cleanup: + if (clien != -1) + close(clien); + if (accept_serv != -1) + close(accept_serv); + if (serv != -1) + close(serv); +} + +static void test_tcp_timewait(struct sock_destroy_prog *skel) +{ + int serv = -1, clien = -1, accept_serv = -1, n; + struct timeval tv = {}; + char buf[1]; + + serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0); + if (!ASSERT_GE(serv, 0, "start_server")) + goto cleanup; + + clien = connect_to_fd(serv, 0); + if (!ASSERT_GE(clien, 0, "connect_to_fd")) + goto cleanup; + + accept_serv = accept(serv, NULL, NULL); + if (!ASSERT_GE(accept_serv, 0, "serv accept")) + goto cleanup; + + /* + * Active close from the client, then close the server side. Once + * recv() sees EOF the server FIN has been processed and the client + * sock is in TIME_WAIT. Block without timeout so a loaded CI box + * can't race us. + */ + if (!ASSERT_OK(setsockopt(clien, SOL_SOCKET, SO_RCVTIMEO, &tv, + sizeof(tv)), "clear rcvtimeo")) + goto cleanup; + if (!ASSERT_OK(shutdown(clien, SHUT_WR), "client shutdown")) + goto cleanup; + + /* + * Make sure the server has seen the client FIN before it closes, + * so the two FINs never cross. + */ + n = recv(accept_serv, buf, sizeof(buf), 0); + if (!ASSERT_EQ(n, 0, "server recv EOF")) + goto cleanup; + + close(accept_serv); + accept_serv = -1; + + /* block until return EOF */ + n = recv(clien, buf, sizeof(buf), 0); + if (!ASSERT_EQ(n, 0, "client recv EOF")) + goto cleanup; + + /* Run iterator program that destroys the timewait client sock. */ + skel->bss->tw_found = 0; + start_iter_sockets(skel->progs.iter_tcp6_timewait); + if (!ASSERT_EQ(skel->bss->tw_found, 1, "timewait sock found")) + goto cleanup; + + ASSERT_OK(skel->bss->tw_destroy_err, "destroy timewait sock"); + + /* The destroyed timewait sock must be gone. */ + skel->bss->tw_found = 0; + start_iter_sockets(skel->progs.iter_tcp6_timewait); + ASSERT_EQ(skel->bss->tw_found, 0, "timewait sock destroyed"); + +cleanup: + if (clien != -1) + close(clien); + if (accept_serv != -1) + close(accept_serv); + if (serv != -1) + close(serv); +} + static void test_udp_client(struct sock_destroy_prog *skel) { int serv = -1, clien = -1, n = 0; @@ -204,6 +321,10 @@ void test_sock_destroy(void) test_tcp_client(skel); if (test__start_subtest("tcp_server")) test_tcp_server(skel); + if (test__start_subtest("tcp_listen_pending")) + test_tcp_listen_pending(skel); + if (test__start_subtest("tcp_timewait")) + test_tcp_timewait(skel); if (test__start_subtest("udp_client")) test_udp_client(skel); if (test__start_subtest("udp_server")) diff --git a/tools/testing/selftests/bpf/prog_tests/spin_lock.c b/tools/testing/selftests/bpf/prog_tests/spin_lock.c index 5c3579438427..e368370262c8 100644 --- a/tools/testing/selftests/bpf/prog_tests/spin_lock.c +++ b/tools/testing/selftests/bpf/prog_tests/spin_lock.c @@ -54,6 +54,8 @@ static struct { { "lock_global_sleepable_helper_subprog", "global function calls are not allowed while holding a lock" }, { "lock_global_sleepable_kfunc_subprog", "global function calls are not allowed while holding a lock" }, { "lock_global_sleepable_subprog_indirect", "global function calls are not allowed while holding a lock" }, + { "callback_value_lock_identity", "bpf_spin_unlock of different lock" }, + { "callback_inner_map_value_lock_identity", "bpf_spin_unlock of different lock" }, }; static int match_regex(const char *pattern, const char *string) diff --git a/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c new file mode 100644 index 000000000000..7acdbd5757a9 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <netinet/tcp.h> + +#include "test_progs.h" +#include "network_helpers.h" +#include "test_tc_change_tail_pmtu.skel.h" + +#define CLIENT_NS "tc-change-tail-cli-ns" +#define SERVER_NS "tc-change-tail-srv-ns" +#define CLIENT_IP "192.168.1.1" +#define SERVER_IP "192.168.1.2" + +#define TEST_PMTU 1000 +#define TEST_MSS_MAX (TEST_PMTU - 20 - 20) +#define TIMEOUT_MS 3000 +#define XFER_BYTES 8192 + +void test_tc_change_tail_pmtu(void) +{ + LIBBPF_OPTS(bpf_tcx_opts, tcx_opts); + int mss_before = 0, mss_after = 0, ifindex, port; + int srv_fd = -1, srv_conn_fd = -1, cli_fd = -1; + struct test_tc_change_tail_pmtu *skel = NULL; + struct nstoken *nstoken = NULL; + static char buf[XFER_BYTES]; + socklen_t optlen; + ssize_t bytes; + size_t total; + + if (!ASSERT_OK(make_netns(CLIENT_NS), "make client ns")) + return; + if (!ASSERT_OK(make_netns(SERVER_NS), "make server ns")) + goto out_client_ns; + + nstoken = open_netns(CLIENT_NS); + if (!ASSERT_OK_PTR(nstoken, "open client ns")) + goto out; + SYS(out, "ip link add veth1 type veth peer name veth2 netns " SERVER_NS); + SYS(out, "ip -4 addr add " CLIENT_IP "/24 dev veth1"); + SYS(out, "ip link set veth1 up"); + ifindex = if_nametoindex("veth1"); + if (!ASSERT_NEQ(ifindex, 0, "if_nametoindex")) + goto out; + close_netns(nstoken); + nstoken = NULL; + + nstoken = open_netns(SERVER_NS); + if (!ASSERT_OK_PTR(nstoken, "open server ns")) + goto out; + SYS(out, "ip -4 addr add " SERVER_IP "/24 dev veth2"); + SYS(out, "ip link set veth2 up"); + srv_fd = start_server(AF_INET, SOCK_STREAM, SERVER_IP, 0, TIMEOUT_MS); + if (!ASSERT_OK_FD(srv_fd, "start server")) + goto out; + close_netns(nstoken); + nstoken = NULL; + + skel = test_tc_change_tail_pmtu__open_and_load(); + if (!ASSERT_OK_PTR(skel, "open and load skeleton")) + goto out; + + port = get_socket_local_port(srv_fd); + if (!ASSERT_GE(port, 0, "get server port")) + goto out; + + skel->bss->server_port = port; + skel->bss->pmtu = TEST_PMTU; + + nstoken = open_netns(CLIENT_NS); + if (!ASSERT_OK_PTR(nstoken, "open client ns")) + goto out; + + skel->links.change_tail_icmp = + bpf_program__attach_tcx(skel->progs.change_tail_icmp, ifindex, + &tcx_opts); + if (!ASSERT_OK_PTR(skel->links.change_tail_icmp, "attach tcx")) + goto out; + + cli_fd = connect_to_fd(srv_fd, TIMEOUT_MS); + if (!ASSERT_OK_FD(cli_fd, "connect to server")) + goto out; + srv_conn_fd = accept(srv_fd, NULL, NULL); + if (!ASSERT_OK_FD(srv_conn_fd, "accept connection")) + goto out; + if (!ASSERT_OK(settimeo(srv_conn_fd, TIMEOUT_MS), "set server timeout")) + goto out; + + optlen = sizeof(mss_before); + if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_before, + &optlen), "get mss before")) + goto out; + + bytes = send(cli_fd, buf, sizeof(buf), 0); + if (!ASSERT_EQ(bytes, (ssize_t)sizeof(buf), "send data")) + goto out; + + for (total = 0; total < sizeof(buf); total += bytes) { + bytes = recv(srv_conn_fd, buf, sizeof(buf), 0); + if (bytes <= 0) + break; + } + + ASSERT_EQ(total, sizeof(buf), "receive data"); + ASSERT_OK(skel->data->change_tail_ret, "change tail"); + ASSERT_OK(skel->bss->adjust_room_ret, "adjust room"); + ASSERT_TRUE(skel->bss->icmp_sent, "icmp sent"); + + optlen = sizeof(mss_after); + if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_after, + &optlen), "get mss after")) + goto out; + + ASSERT_LT(mss_after, mss_before, "mss reduced"); + ASSERT_LE(mss_after, TEST_MSS_MAX, "mss below pmtu"); +out: + close(srv_conn_fd); + close(cli_fd); + close(srv_fd); + test_tc_change_tail_pmtu__destroy(skel); + close_netns(nstoken); + remove_netns(SERVER_NS); +out_client_ns: + remove_netns(CLIENT_NS); +} diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c index 64ac49ad67e6..8b439e194bcc 100644 --- a/tools/testing/selftests/bpf/prog_tests/verifier.c +++ b/tools/testing/selftests/bpf/prog_tests/verifier.c @@ -23,6 +23,7 @@ #include "verifier_bpf_trap.skel.h" #include "verifier_bswap.skel.h" #include "verifier_btf_ctx_access.skel.h" +#include "verifier_btf_flex_array.skel.h" #include "verifier_btf_unreliable_prog.skel.h" #include "verifier_call_large_imm.skel.h" #include "verifier_cfg.skel.h" @@ -53,6 +54,7 @@ #include "verifier_iterating_callbacks.skel.h" #include "verifier_jeq_infer_not_null.skel.h" #include "verifier_jit_convergence.skel.h" +#include "verifier_kfunc_perfmon.skel.h" #include "verifier_ld_ind.skel.h" #include "verifier_ldsx.skel.h" #include "verifier_leak_ptr.skel.h" @@ -186,6 +188,7 @@ void test_verifier_bpf_get_stack(void) { RUN(verifier_bpf_get_stack); } void test_verifier_bpf_trap(void) { RUN(verifier_bpf_trap); } void test_verifier_bswap(void) { RUN(verifier_bswap); } void test_verifier_btf_ctx_access(void) { RUN(verifier_btf_ctx_access); } +void test_verifier_btf_flex_array(void) { RUN(verifier_btf_flex_array); } void test_verifier_btf_unreliable_prog(void) { RUN(verifier_btf_unreliable_prog); } void test_verifier_call_large_imm(void) { RUN(verifier_call_large_imm); } void test_verifier_cfg(void) { RUN(verifier_cfg); } @@ -216,6 +219,7 @@ void test_verifier_int_ptr(void) { RUN(verifier_int_ptr); } void test_verifier_iterating_callbacks(void) { RUN(verifier_iterating_callbacks); } void test_verifier_jeq_infer_not_null(void) { RUN(verifier_jeq_infer_not_null); } void test_verifier_jit_convergence(void) { RUN(verifier_jit_convergence); } +void test_verifier_kfunc_perfmon(void) { RUN(verifier_kfunc_perfmon); } void test_verifier_load_acquire(void) { RUN(verifier_load_acquire); } void test_verifier_ld_ind(void) { RUN(verifier_ld_ind); } void test_verifier_ldsx(void) { RUN(verifier_ldsx); } diff --git a/tools/testing/selftests/bpf/progs/dynptr_success.c b/tools/testing/selftests/bpf/progs/dynptr_success.c index e0745b6e467e..b668ebd61fc7 100644 --- a/tools/testing/selftests/bpf/progs/dynptr_success.c +++ b/tools/testing/selftests/bpf/progs/dynptr_success.c @@ -10,6 +10,7 @@ #include "errno.h" #define PAGE_SIZE_64K 65536 +#define TEST_SKB_LINEAR_SIZE (sizeof(struct ethhdr) + sizeof(struct iphdr)) char _license[] SEC("license") = "GPL"; @@ -212,6 +213,25 @@ int test_dynptr_skb_data(struct __sk_buff *skb) } SEC("?tc") +int test_dynptr_skb_slice_non_linear(struct __sk_buff *skb) +{ + struct bpf_dynptr ptr; + void *data; + + if (bpf_dynptr_from_skb(skb, 0, &ptr)) { + err = 1; + return 1; + } + + /* Ensure we cannot read past the end of the buffer. */ + data = bpf_dynptr_slice(&ptr, TEST_SKB_LINEAR_SIZE + 1, NULL, 1); + if (data) + err = 2; + + return 1; +} + +SEC("?tc") int test_dynptr_skb_meta_data(struct __sk_buff *skb) { struct bpf_dynptr meta; diff --git a/tools/testing/selftests/bpf/progs/exceptions.c b/tools/testing/selftests/bpf/progs/exceptions.c index c8d716fbd419..91c81971e58c 100644 --- a/tools/testing/selftests/bpf/progs/exceptions.c +++ b/tools/testing/selftests/bpf/progs/exceptions.c @@ -212,6 +212,36 @@ int exception_throw_subprog(struct __sk_buff *ctx) return 0; } +u64 exception_cb_stack_src = 0x1234; + +/* + * The address handed to the helper has to be this callback's own stack + * slot, not one from a frame that is already gone. + */ +__noinline int exception_cb_stack(u64 cookie) +{ + volatile u64 val = 0xdead; + + bpf_probe_read_kernel((void *)&val, sizeof(val), &exception_cb_stack_src); + return val; +} + +/* Throws from a subprogram that has a stack of its own. */ +__noinline static int throwing_subprog_stack(struct __sk_buff *ctx) +{ + volatile u64 pad[4] = {}; + + bpf_throw(pad[0]); + return 0; +} + +SEC("tc") +__exception_cb(exception_cb_stack) +int exception_throw_subprog_stack_cb(struct __sk_buff *ctx) +{ + return throwing_subprog_stack(ctx); +} + __noinline int assert_nz_gfunc(u64 c) { volatile u64 cookie = c; diff --git a/tools/testing/selftests/bpf/progs/iters_state_safety.c b/tools/testing/selftests/bpf/progs/iters_state_safety.c index 646026430e9b..e5bb9fe6d5e5 100644 --- a/tools/testing/selftests/bpf/progs/iters_state_safety.c +++ b/tools/testing/selftests/bpf/progs/iters_state_safety.c @@ -52,6 +52,28 @@ int create_and_destroy(void *ctx) return 0; } +/* fp+0 is not a stack slot. bpf_get_spi(0) used to alias spi 0 (fp-8). */ +SEC("?raw_tp") +__failure __msg("cannot pass in iter at an offset=0") +int destroy_fp0_fail(void *ctx) +{ + struct bpf_iter_num iter; + + asm volatile ("r1 = %[iter];" + "r2 = 0;" + "r3 = 1000;" + "call %[bpf_iter_num_new];" + /* r10 is fp+0, one byte above the top of the BPF stack */ + "r1 = r10;" + "call %[bpf_iter_num_destroy];" + : + : __imm_ptr(iter), ITER_HELPERS + : __clobber_common + ); + + return 0; +} + SEC("?raw_tp") __failure __msg("Unreleased reference id=1") int create_and_forget_to_destroy_fail(void *ctx) diff --git a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c index 9e0bf7a54cec..0a8887543218 100644 --- a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c +++ b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c @@ -7,6 +7,8 @@ #include "bpf_tracing_net.h" __be16 serv_port = 0; +int tw_found = 0; +int tw_destroy_err = 0; int bpf_sock_destroy(struct sock_common *sk) __ksym; @@ -100,6 +102,34 @@ int iter_tcp6_server(struct bpf_iter__tcp *ctx) return 0; } +SEC("iter/tcp") +int iter_tcp6_timewait(struct bpf_iter__tcp *ctx) +{ + struct sock_common *sk_common = ctx->sk_common; + __u64 *val; + int key = 0; + + if (!sk_common) + return 0; + + if (sk_common->skc_family != AF_INET6) + return 0; + + if (!bpf_skc_to_tcp_timewait_sock(sk_common)) + return 0; + + val = bpf_map_lookup_elem(&tcp_conn_sockets, &key); + if (!val) + return 0; + /* The timewait sock inherits the cookie of the closed client sock. */ + if (bpf_get_socket_cookie(sk_common) != *val) + return 0; + + tw_found++; + tw_destroy_err = bpf_sock_destroy(sk_common); + + return 0; +} SEC("iter/udp") int iter_udp6_client(struct bpf_iter__udp *ctx) diff --git a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c index f678ee6bd7ea..55282f20fa32 100644 --- a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c +++ b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c @@ -14,17 +14,18 @@ struct array_map { __type(key, int); __type(value, struct foo); __uint(max_entries, 1); -} array_map SEC(".maps"); +} array_map SEC(".maps"), array_map_b SEC(".maps"); struct { __uint(type, BPF_MAP_TYPE_ARRAY_OF_MAPS); - __uint(max_entries, 1); + __uint(max_entries, 2); __type(key, int); __type(value, int); __array(values, struct array_map); } map_of_maps SEC(".maps") = { .values = { [0] = &array_map, + [1] = &array_map_b, }, }; @@ -314,4 +315,66 @@ int lock_global_sleepable_subprog_indirect(struct __sk_buff *ctx) return ret; } +struct { + __uint(type, BPF_MAP_TYPE_ARRAY); + __uint(max_entries, 2); + __type(key, int); + __type(value, struct foo); +} callback_array_map SEC(".maps"); + +struct callback_ctx { + struct foo *value; +}; + +static long lock_different_value(struct bpf_map *map, int *key, + struct foo *value, struct callback_ctx *ctx) +{ + bpf_spin_lock(&value->lock); + bpf_spin_unlock(&ctx->value->lock); + return 0; +} + +static long nest_lock_different_value(struct bpf_map *map, int *key, + struct foo *value, void *data) +{ + struct callback_ctx ctx = { .value = value }; + + bpf_for_each_map_elem(&callback_array_map, lock_different_value, &ctx, 0); + return 0; +} + +SEC("?tc") +int callback_value_lock_identity(void *ctx) +{ + bpf_for_each_map_elem(&callback_array_map, nest_lock_different_value, NULL, 0); + return 0; +} + +static long nest_lock_different_inner_value(struct bpf_map *map, int *key, + struct foo *value, void *data) +{ + struct callback_ctx ctx = { .value = value }; + int inner_key = 1; + void *inner_map; + + inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key); + if (!inner_map) + return 0; + bpf_for_each_map_elem(inner_map, lock_different_value, &ctx, 0); + return 0; +} + +SEC("?tc") +int callback_inner_map_value_lock_identity(void *ctx) +{ + int inner_key = 0; + void *inner_map; + + inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key); + if (!inner_map) + return 0; + bpf_for_each_map_elem(inner_map, nest_lock_different_inner_value, NULL, 0); + return 0; +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c new file mode 100644 index 000000000000..5c4c07545bc9 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <stdbool.h> +#include <stddef.h> + +#include <linux/bpf.h> +#include <linux/icmp.h> +#include <linux/if_ether.h> +#include <linux/in.h> +#include <linux/ip.h> +#include <linux/tcp.h> + +#include <bpf/bpf_helpers.h> +#include <bpf/bpf_endian.h> + +#define ICMP_SAMPLE_LEN (sizeof(struct iphdr) + 8) +#define ICMP_HDRS_LEN (sizeof(struct iphdr) + sizeof(struct icmphdr)) + +__be16 server_port = 0; +__u16 pmtu = 0; + +long change_tail_ret = 1; +long adjust_room_ret = 0; +bool icmp_sent = false; +bool icmp_err = false; + +static __always_inline __sum16 csum_fold(__wsum csum) +{ + csum = (csum & 0xffff) + (csum >> 16); + csum = (csum & 0xffff) + (csum >> 16); + + return (__sum16)~csum; +} + +SEC("tc/egress") +int change_tail_icmp(struct __sk_buff *skb) +{ + __u8 smac[ETH_ALEN], dmac[ETH_ALEN]; + void *data, *data_end; + struct icmphdr *icmp; + struct ethhdr *eth; + struct tcphdr *tcp; + __be32 saddr, daddr; + struct iphdr *ip; + __wsum csum; + + if (icmp_sent || icmp_err) + return TCX_PASS; + + data = (void *)(long)skb->data; + data_end = (void *)(long)skb->data_end; + + eth = data; + if ((void *)(eth + 1) > data_end) + return TCX_PASS; + if (eth->h_proto != bpf_htons(ETH_P_IP)) + return TCX_PASS; + + ip = (void *)(eth + 1); + if ((void *)(ip + 1) > data_end) + return TCX_PASS; + if (ip->ihl != 5 || ip->protocol != IPPROTO_TCP) + return TCX_PASS; + + tcp = (void *)(ip + 1); + if ((void *)(tcp + 1) > data_end) + return TCX_PASS; + if (tcp->dest != server_port) + return TCX_PASS; + if (bpf_ntohs(ip->tot_len) <= sizeof(*ip) + tcp->doff * 4) + return TCX_PASS; + + __builtin_memcpy(smac, eth->h_source, ETH_ALEN); + __builtin_memcpy(dmac, eth->h_dest, ETH_ALEN); + saddr = ip->saddr; + daddr = ip->daddr; + + change_tail_ret = bpf_skb_change_tail(skb, ETH_HLEN + ICMP_SAMPLE_LEN, 0); + if (change_tail_ret) { + icmp_err = true; + return TCX_PASS; + } + + adjust_room_ret = bpf_skb_adjust_room(skb, ICMP_HDRS_LEN, + BPF_ADJ_ROOM_MAC, + BPF_F_ADJ_ROOM_NO_CSUM_RESET); + if (adjust_room_ret) { + icmp_err = true; + return TCX_DROP; + } + + data = (void *)(long)skb->data; + data_end = (void *)(long)skb->data_end; + + eth = data; + ip = (void *)(eth + 1); + icmp = (void *)(ip + 1); + if ((void *)icmp + sizeof(*icmp) + ICMP_SAMPLE_LEN > data_end) { + icmp_err = true; + return TCX_DROP; + } + + __builtin_memcpy(eth->h_dest, smac, ETH_ALEN); + __builtin_memcpy(eth->h_source, dmac, ETH_ALEN); + + __builtin_memset(icmp, 0, sizeof(*icmp)); + icmp->type = ICMP_DEST_UNREACH; + icmp->code = ICMP_FRAG_NEEDED; + icmp->un.frag.mtu = bpf_htons(pmtu); + + __builtin_memset(ip, 0, sizeof(*ip)); + ip->version = 4; + ip->ihl = 5; + ip->ttl = 64; + ip->protocol = IPPROTO_ICMP; + ip->tot_len = bpf_htons(ICMP_HDRS_LEN + ICMP_SAMPLE_LEN); + ip->saddr = daddr; + ip->daddr = saddr; + + csum = bpf_csum_diff(NULL, 0, (__be32 *)icmp, + sizeof(*icmp) + ICMP_SAMPLE_LEN, 0); + icmp->checksum = csum_fold(csum); + csum = bpf_csum_diff(NULL, 0, (__be32 *)ip, sizeof(*ip), 0); + ip->check = csum_fold(csum); + icmp_sent = true; + return bpf_redirect(skb->ifindex, BPF_F_INGRESS); +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_arena.c b/tools/testing/selftests/bpf/progs/verifier_arena.c index 815f342eb4b0..3e33766547c0 100644 --- a/tools/testing/selftests/bpf/progs/verifier_arena.c +++ b/tools/testing/selftests/bpf/progs/verifier_arena.c @@ -562,6 +562,55 @@ int arena_ptr_add_arena_ptr(void *ctx) } SEC("syscall") +__failure __msg("same insn cannot be used with and without arena pointer") +int mixed_arena_scalar_alu64_scalar_first(void *ctx) +{ + volatile register __u64 reg asm("r3"); + __u32 pick_arena = bpf_get_prandom_u32(); + + reg = 1ULL << 32; + + if (pick_arena) { + asm volatile ( + "r9 = %[arena] ll;" + "%[reg] = 0;" + "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);" + : [reg] "=r"(reg) + : __imm_addr(arena) + : "r9" + ); + } + + reg += 1; + + return 0; +} + +SEC("syscall") +__failure __msg("same insn cannot be used with and without arena pointer") +int mixed_arena_scalar_alu64_arena_first(void *ctx) +{ + volatile register __u64 reg asm("r3"); + __u32 pick_scalar = bpf_get_prandom_u32(); + + asm volatile ( + "r9 = %[arena] ll;" + "%[reg] = 0;" + "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);" + : [reg] "=r"(reg) + : __imm_addr(arena) + : "r9" + ); + + if (pick_scalar) + reg = 1ULL << 32; + + reg += 1; + + return 0; +} + +SEC("syscall") __success __retval(0) int scalar_xor_arena_ptr(void *ctx) { diff --git a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c index e0926767bbd3..1b653bfb63eb 100644 --- a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c +++ b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c @@ -9,6 +9,11 @@ char _license[] SEC("license") = "GPL"; +struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym; +void bpf_task_release(struct task_struct *p) __ksym; +void bpf_rcu_read_lock(void) __ksym; +void bpf_rcu_read_unlock(void) __ksym; + /* Timer tests */ struct timer_elem { @@ -164,6 +169,7 @@ int syscall_btf_find_prog(void *ctx) struct wq_elem { struct bpf_wq w; + struct task_struct __kptr *task; }; struct { @@ -217,6 +223,106 @@ int wq_sleepable_prog(void *ctx) return 0; } +__noinline int wq_global_acquire(void) +{ + struct task_struct *task, *acquired; + struct wq_elem *val; + int key = 0; + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + task = val->task; + if (!task) + return 0; + + acquired = bpf_task_acquire(task); + if (acquired) + bpf_task_release(acquired); + return 0; +} + +static int wq_global_rcu_cb(void *map, int *key, void *value) +{ + wq_global_acquire(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__failure __msg("R1 must be a rcu pointer") +int wq_global_rcu_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_rcu_cb, 0); + return 0; +} + +static int wq_global_rcu_lock_cb(void *map, int *key, void *value) +{ + bpf_rcu_read_lock(); + wq_global_acquire(); + bpf_rcu_read_unlock(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__success +int wq_global_rcu_lock_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + /* Verify the same global subprog in non-sleepable and protected contexts. */ + wq_global_acquire(); + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_rcu_lock_cb, 0); + return 0; +} + +__weak __noinline int wq_global_no_rcu(void) +{ + return 0; +} + +static int wq_global_no_rcu_cb(void *map, int *key, void *value) +{ + wq_global_no_rcu(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__success __log_level(4) +__msg("subprog {{[0-9]+}} (wq_global_no_rcu) global insns_self 4 insns_total 4 stack 0") +int wq_global_no_rcu_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + /* Verify the same global in non-sleepable and unprotected contexts. */ + wq_global_no_rcu(); + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_no_rcu_cb, 0); + return 0; +} + /* Task work tests */ struct task_work_elem { diff --git a/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c new file mode 100644 index 000000000000..59b84261f622 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <vmlinux.h> +#include <bpf/bpf_helpers.h> + +#include "bpf_experimental.h" +#include "bpf_misc.h" + +struct test_empty_event {}; + +struct test_flex_batch { + int nr; + struct test_empty_event events[]; +}; + +struct map_value { + struct test_flex_batch __kptr *batch; +}; + +struct { + __uint(type, BPF_MAP_TYPE_ARRAY); + __type(key, int); + __type(value, struct map_value); + __uint(max_entries, 1); +} batches SEC(".maps"); + +SEC("syscall") +__description("btf walk into flexible array of zero-sized elements") +__failure __msg("access beyond struct test_flex_batch at off 4 size 1") +int stash_and_peek(void *ctx) +{ + struct test_flex_batch *b, *old; + struct map_value *v; + int key = 0; + + v = bpf_map_lookup_elem(&batches, &key); + if (!v) + return 0; + + b = bpf_obj_new(struct test_flex_batch); + if (!b) + return 0; + b->nr = 1; + + old = bpf_kptr_xchg(&v->batch, b); + if (old) + bpf_obj_drop(old); + + b = v->batch; + if (!b) + return 0; + + return b->nr + *(char *)&b->events[0]; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c index a3d2af8dc839..dcc2dd46751a 100644 --- a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c +++ b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c @@ -350,4 +350,57 @@ int anything_to_untrusted_mem(void *ctx) return 0; } +struct pkt_arg { + __u64 x; + __u8 pad[56]; +}; + +__weak int subprog_pkt_ptr_no_change(struct pkt_arg *p) +{ + if (!p) + return 0; + + return p->x; +} + +SEC("?tc") +__success +int pkt_ptr_to_global_mem_arg_no_change(struct __sk_buff *skb) +{ + void *data = (void *)(long)skb->data; + void *data_end = (void *)(long)skb->data_end; + struct pkt_arg *p = data; + + if ((void *)(p + 1) > data_end) + return 0; + + return subprog_pkt_ptr_no_change(p); +} + +__weak int subprog_pkt_ptr_changes_data(struct __sk_buff *skb __arg_ctx, + struct pkt_arg *p) +{ + if (!p) + return 0; + + bpf_skb_pull_data(skb, 0); + return p->x; +} + +SEC("?tc") +__failure __log_level(2) +__msg("R2 is a packet pointer, but func#{{[0-9]+}} may change packet data") +__msg("Caller passes invalid args into func#{{[0-9]+}} ('subprog_pkt_ptr_changes_data')") +int pkt_ptr_to_global_mem_arg_changes_data(struct __sk_buff *skb) +{ + void *data = (void *)(long)skb->data; + void *data_end = (void *)(long)skb->data_end; + struct pkt_arg *p = data; + + if ((void *)(p + 1) > data_end) + return 0; + + return subprog_pkt_ptr_changes_data(skb, p); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_gotox.c b/tools/testing/selftests/bpf/progs/verifier_gotox.c index 5b18c9a27717..0e27c2c79c57 100644 --- a/tools/testing/selftests/bpf/progs/verifier_gotox.c +++ b/tools/testing/selftests/bpf/progs/verifier_gotox.c @@ -47,6 +47,54 @@ DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_src_reg, BPF_REG_1, 0, 0, __fa DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_off, BPF_REG_0, 1, 0, __failure __msg("BPF_JA|BPF_X uses reserved fields")) DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_imm, BPF_REG_0, 0, 1, __failure __msg("BPF_JA|BPF_X uses reserved fields")) +#define DEFINE_TERMINAL_GOTOX_PROG(NAME, BASE) \ + __naked void NAME(void) \ + { \ + asm volatile (" \ + .pushsection .jumptables,\"\",@progbits; \ +jt0_%=: \ + .quad ret0_%= - " BASE "; \ + .size jt0_%=, 8; \ + .global jt0_%=; \ + .popsection; \ + \ + r0 = jt0_%= ll; \ + r0 = *(u64 *)(r0 + 0); \ + goto end_%=; \ +ret0_%=: \ + r0 = 0; \ + exit; \ +end_%=: \ + .8byte %[gotox_r0]; \ +" : \ + : __imm_insn(gotox_r0, BPF_RAW_INSN(BPF_JMP | BPF_JA | BPF_X, \ + BPF_REG_0, 0, 0, 0)) \ + : __clobber_all); \ + } + +SEC("socket") +__success __retval(0) +DEFINE_TERMINAL_GOTOX_PROG(jump_table_terminal_gotox, "socket") + +static __noinline __used +DEFINE_TERMINAL_GOTOX_PROG(terminal_gotox_subprog1, ".text") + +static __noinline __used int terminal_gotox_subprog2(void) +{ + return 0; +} + +SEC("socket") +__success __retval(0) +__naked void jump_table_terminal_gotox_subprog(void) +{ + asm volatile (" \ + call terminal_gotox_subprog1; \ + call terminal_gotox_subprog2; \ + exit; \ +" ::: __clobber_all); +} + /* * Gotox is forbidden when there is no jump table loaded * which points to the sub-function where the gotox is used diff --git a/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c new file mode 100644 index 000000000000..76c39ef30e96 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <vmlinux.h> +#include <bpf/bpf_helpers.h> +#include "bpf_misc.h" + +void *user_ptr; +char dynptr_buf[8]; +u64 kaddr; + +extern struct kmem_cache *bpf_get_kmem_cache(u64 addr) __ksym; + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_rdonly_cast is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int rdonly_cast_noperfmon(void *ctx) +{ + char *p = bpf_rdonly_cast(0, 0); + + return p[0x7fff]; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_probe_read_kernel_dynptr is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int probe_read_kernel_dynptr_noperfmon(void *ctx) +{ + struct bpf_dynptr dptr; + + bpf_dynptr_from_mem(dynptr_buf, sizeof(dynptr_buf), 0, &dptr); + bpf_probe_read_kernel_dynptr(&dptr, 0, sizeof(dynptr_buf), user_ptr); + return 0; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_stream_vprintk is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int stream_vprintk_noperfmon(void *ctx) +{ + bpf_stream_printk(BPF_STDOUT, "%pB", (void *)kaddr); + return 0; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_get_kmem_cache is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int get_kmem_cache_noperfmon(void *ctx) +{ + return !!bpf_get_kmem_cache(kaddr); +} + +__weak int subprog_untrusted_read(void *p __arg_untrusted) +{ + return *(char *)p; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("rdonly_untrusted_mem access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int arg_untrusted_read_noperfmon(void *ctx) +{ + return subprog_untrusted_read(0); +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_loops1.c b/tools/testing/selftests/bpf/progs/verifier_loops1.c index d248ce877f14..48a966cda199 100644 --- a/tools/testing/selftests/bpf/progs/verifier_loops1.c +++ b/tools/testing/selftests/bpf/progs/verifier_loops1.c @@ -303,4 +303,40 @@ __naked void maybe_exit_scc_bug1(void) ::: __clobber_all); } +/* + * The loop reads zero from the caller's stack on its first iteration and + * one from the callee's stack on its second iteration. At the loop header, + * only the frame number of the pointer in r1 changes. + */ +static __naked __noinline __used +void loop_stack_frames_reg(void) +{ + asm volatile ( + "*(u64 *)(r10 - 8) = 1;" +"1:" + "r0 = *(u64 *)(r1 + 0);" + "if r0 != 0 goto 2f;" + "r1 = r10;" + "r1 += -8;" + "goto 1b;" +"2:" + "exit;" + ::: __clobber_all); +} + +SEC("xdp") +__description("bounded loop changing stack frame in a register") +__success __retval(1) +__flag(BPF_F_TEST_STATE_FREQ) +__naked void bounded_loop_stack_frames_reg(void) +{ + asm volatile ( + "*(u64 *)(r10 - 8) = 0;" + "r1 = r10;" + "r1 += -8;" + "call loop_stack_frames_reg;" + "exit;" + ::: __clobber_all); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_sock.c b/tools/testing/selftests/bpf/progs/verifier_sock.c index 4f2f3209eec8..bf9f6fb6582c 100644 --- a/tools/testing/selftests/bpf/progs/verifier_sock.c +++ b/tools/testing/selftests/bpf/progs/verifier_sock.c @@ -88,6 +88,44 @@ l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_family]); \ : __clobber_all); } +SEC("socket") +__description("skb->sk: sk->rx_queue_mapping [no sign extension]") +__success __success_unpriv __retval(0) +__naked void sk_rx_queue_mapping_no_sign_ext(void) +{ + asm volatile (" \ + r1 = *(u64*)(r1 + %[__sk_buff_sk]); \ + if r1 != 0 goto l0_%=; \ + r0 = 0xdead; \ + exit; \ +l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_rx_queue_mapping]); \ + r0 >>= 32; \ + exit; \ +" : + : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)), + __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping)) + : __clobber_all); +} + +SEC("socket") +__description("skb->sk: sk->rx_queue_mapping [narrow load mask]") +__success __success_unpriv __retval(0) +__naked void sk_rx_queue_mapping_narrow_load_mask(void) +{ + asm volatile (" \ + r1 = *(u64*)(r1 + %[__sk_buff_sk]); \ + if r1 != 0 goto l0_%=; \ + r0 = 0xdead; \ + exit; \ +l0_%=: r0 = *(u16*)(r1 + %[bpf_sock_rx_queue_mapping]); \ + r0 >>= 16; \ + exit; \ +" : + : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)), + __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping)) + : __clobber_all); +} + SEC("cgroup/skb") __description("skb->sk: sk->type [fullsock field]") __failure __msg("invalid sock_common access") diff --git a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c index 0b86d95a4133..9866bc154194 100644 --- a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c +++ b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c @@ -1719,4 +1719,39 @@ l0_%=: r0 = 0; \ : __clobber_all); } +SEC("xdp") +__description("XDP pkt regsafe preserves packet pointer class displacement") +__failure __msg("R2 min value is outside of the allowed memory range") +__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ) +__naked void pkt_regsafe_class_displacement(void) +{ + asm volatile (" \ + r8 = *(u32 *)(r1 + %[xdp_md_data_end]); \ + r9 = *(u32 *)(r1 + %[xdp_md_data]); \ + r4 = *(u32 *)(r1 + %[xdp_md_rx_queue_index]); \ + r4 &= 15; \ + r0 = *(u32 *)(r1 + %[xdp_md_ingress_ifindex]); \ + if r0 != 0 goto l0_%=; \ + r2 = r9; \ + r2 += r4; \ + r3 = r2; \ + r3 += 8; \ + goto l1_%=; \ +l0_%=: r4 &= 3; \ + r4 += 8; \ + r2 = r9; \ + r2 += r4; \ + r3 = r2; \ +l1_%=: if r3 > r8 goto l2_%=; \ + r0 = *(u64 *)(r2 + 0); \ +l2_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(xdp_md_data, offsetof(struct xdp_md, data)), + __imm_const(xdp_md_data_end, offsetof(struct xdp_md, data_end)), + __imm_const(xdp_md_rx_queue_index, offsetof(struct xdp_md, rx_queue_index)), + __imm_const(xdp_md_ingress_ifindex, offsetof(struct xdp_md, ingress_ifindex)) + : __clobber_all); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/cgroup/test_cpuset_prs.sh b/tools/testing/selftests/cgroup/test_cpuset_prs.sh index 131d8b4551ef..7efd5e645767 100755 --- a/tools/testing/selftests/cgroup/test_cpuset_prs.sh +++ b/tools/testing/selftests/cgroup/test_cpuset_prs.sh @@ -298,6 +298,8 @@ TEST_MATRIX=( " C0-4:X2-4 C1-4:X2-4:P2 C2-4:X4:P1 \ . . . X1 . 0 A1:0-1|A2:2-4|A3:2-4 \ A1:P0|A2:P2|A3:P-1 2-4" + " CX1-3:P1 CX1-3 CX1-3 . . . P1 . 0 A1:1-3|A2:1-3|A3:1-3 \ + A1:P1|A2:P0|A3:P-1" # Remote partition offline tests " C0-3 C1-3 C2-3 . X2-3 X2-3 X2-3:P2:O2=0 . 0 A1:0-1|A2:1|A3:3 A1:P0|A3:P2 2-3" diff --git a/tools/testing/selftests/cgroup/test_memcontrol.c b/tools/testing/selftests/cgroup/test_memcontrol.c index 3a84d068fbf3..0ed82347044e 100644 --- a/tools/testing/selftests/cgroup/test_memcontrol.c +++ b/tools/testing/selftests/cgroup/test_memcontrol.c @@ -30,7 +30,7 @@ static int page_size; int get_temp_fd(void) { - return open(".", O_TMPFILE | O_RDWR | O_EXCL); + return open(".", O_TMPFILE | O_RDWR | O_EXCL, 0600); } int alloc_pagecache(int fd, size_t size) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 609c48f38524..8df54b59513a 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -20,6 +20,7 @@ static int page_size; #define PATH_ZSWAP "/sys/module/zswap" #define PATH_ZSWAP_ENABLED "/sys/module/zswap/parameters/enabled" +#define PATH_ZSWAP_SHRINKER_ENABLED "/sys/module/zswap/parameters/shrinker_enabled" #define PATH_ZSWAP_STORED_PAGES "/sys/kernel/debug/zswap/stored_pages" static int read_int(const char *path, size_t *value) @@ -444,6 +445,16 @@ static int test_zswap_writeback_disabled(const char *root) return test_zswap_writeback(root, false); } +static bool zswap_shrinker_enabled(void) +{ + char value[2]; + + if (read_text(PATH_ZSWAP_SHRINKER_ENABLED, value, sizeof(value)) <= 0) + return 0; + + return value[0] == 'Y'; +} + /* * When trying to store a memcg page in zswap, if the memcg hits its memory * limit in zswap, writeback should affect only the zswapped pages of that @@ -453,6 +464,7 @@ static int test_no_invasive_cgroup_shrink(const char *root) { int ret = KSFT_FAIL; unsigned int off; + long zswpwb_before, zswpwb_after, zswpwb_target; size_t allocation_size = page_size * 1024; unsigned int nr_pages = allocation_size / page_size; char zswap_max_buf[32], mem_max_buf[32]; @@ -488,6 +500,14 @@ static int test_no_invasive_cgroup_shrink(const char *root) if (cg_read_key_long(zw_group, "memory.stat", "zswapped") < 1) goto out; + /* If the shrinker is enabled, try to let the writebacks finish first */ + if (zswap_shrinker_enabled()) + sleep(5); + + zswpwb_before = get_cg_wb_count(zw_group); + if (zswpwb_before < 0) + goto out; + /* Push wb_group memory into zswap with hard-to-compress data to trigger wb */ if (cg_enter_current(wb_group)) goto out; @@ -500,9 +520,13 @@ static int test_no_invasive_cgroup_shrink(const char *root) getrandom(&wb_allocation[off], page_size/4, 0); } - /* Verify that only zswapped memory from gwb_group has been written back */ - if (wait_for_writeback(wb_group, 5000) > 0 && get_cg_wb_count(zw_group) == 0) + /* Verify that only zswapped memory from wb_group has been written back */ + zswpwb_target = wait_for_writeback(wb_group, 5000); + zswpwb_after = get_cg_wb_count(zw_group); + + if (zswpwb_target > 0 && zswpwb_before == zswpwb_after) ret = KSFT_PASS; + out: cg_enter_current(root); if (zw_group) { diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore b/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore deleted file mode 100644 index 493fbcf8d9ec..000000000000 --- a/tools/testing/selftests/filesystems/mntns_cleanup/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -mntns_cleanup_test diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/Makefile b/tools/testing/selftests/filesystems/mntns_cleanup/Makefile deleted file mode 100644 index 0e09e7030a5c..000000000000 --- a/tools/testing/selftests/filesystems/mntns_cleanup/Makefile +++ /dev/null @@ -1,6 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0 -TEST_GEN_PROGS := mntns_cleanup_test - -CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) - -include ../../lib.mk diff --git a/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c b/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c deleted file mode 100644 index 5209712568b1..000000000000 --- a/tools/testing/selftests/filesystems/mntns_cleanup/mntns_cleanup_test.c +++ /dev/null @@ -1,58 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0 - -#define _GNU_SOURCE -#include <errno.h> -#include <fcntl.h> -#include <sched.h> -#include <sys/mount.h> -#include <sys/stat.h> -#include <unistd.h> - -#include "../../kselftest_harness.h" - -FIXTURE(mntns_cleanup) { -}; - -FIXTURE_SETUP(mntns_cleanup) -{ - if (geteuid() != 0) - SKIP(return, "test requires CAP_SYS_ADMIN"); - - ASSERT_EQ(unshare(CLONE_NEWNS), 0); - ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); - - rmdir("/mnt_dir"); - ASSERT_EQ(mkdir("/mnt_dir", 0755), 0); - ASSERT_EQ(mount("tmpfs", "/mnt_dir", "tmpfs", 0, NULL), 0); - ASSERT_EQ(mkdir("/mnt_dir/hidden", 0755), 0); - ASSERT_EQ(mkdir("/mnt_dir/hidden/secret", 0755), 0); - ASSERT_EQ(mount("tmpfs", "/mnt_dir/hidden", "tmpfs", 0, NULL), 0); -} - -FIXTURE_TEARDOWN(mntns_cleanup) -{ -} - -/* Mounts must stay connected when a mount namespace is cleaned up. */ -TEST_F(mntns_cleanup, keeps_mounts_connected) -{ - int fd, sfd, err; - - fd = open("/mnt_dir", O_PATH | O_DIRECTORY | O_CLOEXEC); - ASSERT_GE(fd, 0); - - /* Destroy the namespace; the fd keeps /mnt_dir alive. */ - ASSERT_EQ(unshare(CLONE_NEWNS), 0); - - sfd = openat(fd, "hidden/secret", O_RDONLY); - err = errno; - if (sfd >= 0) - close(sfd); - close(fd); - - ASSERT_LT(sfd, 0) - TH_LOG("mount namespace teardown revealed what the overmount covered"); - ASSERT_EQ(err, ENOENT); -} - -TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc index bc9514428dba..07b1177c1634 100644 --- a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc +++ b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc @@ -6,7 +6,7 @@ SYMBOL='name_show' # We skip this test on kernel where SYMBOL is unique or does not exist. -if [ "$(grep -c -E "[[:alnum:]]+ t ${SYMBOL}" /proc/kallsyms)" -le '1' ]; then +if [ "$(grep -c -E "[[:alnum:]]+ t ${SYMBOL}$" /proc/kallsyms)" -le '1' ]; then exit_unsupported fi diff --git a/tools/testing/selftests/hid/hid_bpf.c b/tools/testing/selftests/hid/hid_bpf.c index 7ab86296ff23..32d81ba15a25 100644 --- a/tools/testing/selftests/hid/hid_bpf.c +++ b/tools/testing/selftests/hid/hid_bpf.c @@ -5,7 +5,7 @@ #include <bpf/bpf.h> struct hid_hw_request_syscall_args { - __u8 data[10]; + __u8 data[MAX_BUF_SIZE]; unsigned int hid; int retval; size_t size; @@ -54,11 +54,27 @@ FIXTURE_TEARDOWN(hid_bpf) { hid_bpf_teardown(_metadata, self, variant); \ } while (0) +FIXTURE_VARIANT(hid_bpf) { + __u8 *rdesc; + size_t rdesc_size; +}; + +FIXTURE_VARIANT_ADD(hid_bpf, numbered) { + .rdesc = rdesc, + .rdesc_size = sizeof(rdesc), +}; + +FIXTURE_VARIANT_ADD(hid_bpf, unnumbered) { + .rdesc = fido2_rdesc, + .rdesc_size = sizeof(fido2_rdesc), +}; + FIXTURE_SETUP(hid_bpf) { int err; - err = setup_uhid(_metadata, &self->hid, BUS_USB, 0x0001, 0x0a36, rdesc, sizeof(rdesc)); + err = setup_uhid(_metadata, &self->hid, BUS_USB, 0x0001, 0x0a36, + variant->rdesc, variant->rdesc_size); ASSERT_OK(err); } @@ -175,7 +191,7 @@ TEST_F(hid_bpf, raw_event) const struct test_program progs[] = { { .name = "hid_first_event" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -226,7 +242,7 @@ TEST_F(hid_bpf, subprog_raw_event) const struct test_program progs[] = { { .name = "hid_subprog_first_event" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -284,7 +300,7 @@ TEST_F(hid_bpf, test_attach_detach) { .name = "hid_second_event" }, }; struct bpf_link *link; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err, link_fd; LOAD_PROGRAMS(progs); @@ -369,7 +385,7 @@ TEST_F(hid_bpf, test_hid_change_report) const struct test_program progs[] = { { .name = "hid_change_report_id" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -396,21 +412,24 @@ TEST_F(hid_bpf, test_hid_user_input_report_call) { struct hid_hw_request_syscall_args args = { .retval = -1, - .size = 10, + .size = MAX_BUF_SIZE, }; DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs, .ctx_in = &args, .ctx_size_in = sizeof(args), ); - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err, prog_fd; LOAD_BPF; args.hid = self->hid.hid_id; args.data[0] = 1; /* report ID */ - args.data[1] = 2; /* report ID */ - args.data[2] = 42; /* report ID */ + args.data[1] = 2; + args.data[2] = 42; + + if (variant->rdesc == fido2_rdesc) + args.data[0] = 0; prog_fd = bpf_program__fd(self->skel->progs.hid_user_input_report); @@ -428,8 +447,13 @@ TEST_F(hid_bpf, test_hid_user_input_report_call) /* read the data from hidraw */ memset(buf, 0, sizeof(buf)); err = read(self->hidraw_fd, buf, sizeof(buf)); - ASSERT_EQ(err, 6) TH_LOG("read_hidraw"); - ASSERT_EQ(buf[0], 1); + if (variant->rdesc == rdesc) { + ASSERT_EQ(err, 6) TH_LOG("read_hidraw"); + } else { + ASSERT_EQ(err, 64) + TH_LOG("read_hidraw"); + } + ASSERT_EQ(buf[0], args.data[0]); ASSERT_EQ(buf[1], 2); ASSERT_EQ(buf[2], 42); } @@ -442,7 +466,7 @@ TEST_F(hid_bpf, test_hid_user_output_report_call) { struct hid_hw_request_syscall_args args = { .retval = -1, - .size = 10, + .size = MAX_BUF_SIZE, }; DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs, .ctx_in = &args, @@ -455,8 +479,11 @@ TEST_F(hid_bpf, test_hid_user_output_report_call) args.hid = self->hid.hid_id; args.data[0] = 1; /* report ID */ - args.data[1] = 2; /* report ID */ - args.data[2] = 42; /* report ID */ + args.data[1] = 2; + args.data[2] = 42; + + if (variant->rdesc == fido2_rdesc) + args.data[0] = 0; prog_fd = bpf_program__fd(self->skel->progs.hid_user_output_report); @@ -472,9 +499,14 @@ TEST_F(hid_bpf, test_hid_user_output_report_call) ASSERT_OK(err) TH_LOG("error while calling bpf_prog_test_run_opts"); ASSERT_OK(cond_err) TH_LOG("error while calling waiting for the condition"); - ASSERT_EQ(args.retval, 3); + if (variant->rdesc == rdesc) { + ASSERT_EQ(args.retval, 3); + } else if (variant->rdesc == fido2_rdesc) { + ASSERT_EQ(args.retval, 65) + TH_LOG("report size error, should have 64 + 1 extra byte for the report ID 0"); + } - ASSERT_EQ(output_report[0], 1); + ASSERT_EQ(output_report[0], args.data[0]); ASSERT_EQ(output_report[1], 2); ASSERT_EQ(output_report[2], 42); @@ -491,7 +523,7 @@ TEST_F(hid_bpf, test_hid_user_raw_request_call) .retval = -1, .type = HID_FEATURE_REPORT, .request_type = HID_REQ_GET_REPORT, - .size = 10, + .size = MAX_BUF_SIZE, }; DECLARE_LIBBPF_OPTS(bpf_test_run_opts, tattrs, .ctx_in = &args, @@ -524,7 +556,7 @@ TEST_F(hid_bpf, test_hid_filter_raw_request_call) const struct test_program progs[] = { { .name = "hid_test_filter_raw_request" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -577,7 +609,7 @@ TEST_F(hid_bpf, test_hid_change_raw_request_call) const struct test_program progs[] = { { .name = "hid_test_hidraw_raw_request" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -603,7 +635,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_raw_request_call) const struct test_program progs[] = { { .name = "hid_test_infinite_loop_raw_request" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -626,7 +658,7 @@ TEST_F(hid_bpf, test_hid_filter_output_report_call) const struct test_program progs[] = { { .name = "hid_test_filter_output_report" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -679,7 +711,7 @@ TEST_F(hid_bpf, test_hid_change_output_report_call) const struct test_program progs[] = { { .name = "hid_test_hidraw_output_report" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -703,7 +735,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_output_report_call) const struct test_program progs[] = { { .name = "hid_test_infinite_loop_output_report" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -729,7 +761,7 @@ TEST_F(hid_bpf, test_multiply_events_wq) const struct test_program progs[] = { { .name = "hid_test_multiply_events_wq" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -767,7 +799,7 @@ TEST_F(hid_bpf, test_multiply_events) const struct test_program progs[] = { { .name = "hid_test_multiply_events" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -801,7 +833,7 @@ TEST_F(hid_bpf, test_hid_infinite_loop_input_report_call) const struct test_program progs[] = { { .name = "hid_test_infinite_loop_input_report" }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -855,7 +887,7 @@ TEST_F(hid_bpf, test_hid_attach_flags) .insert_head = 0, }, }; - __u8 buf[10] = {0}; + __u8 buf[MAX_BUF_SIZE] = {0}; int err; LOAD_PROGRAMS(progs); @@ -886,6 +918,9 @@ TEST_F(hid_bpf, test_rdesc_fixup) }; int err, desc_size; + if (variant->rdesc != rdesc) + SKIP(return, "not compatible report descriptor"); + LOAD_PROGRAMS(progs); /* check that hid_rdesc_fixup() was executed */ diff --git a/tools/testing/selftests/hid/hid_common.h b/tools/testing/selftests/hid/hid_common.h index e3b267446fa0..b7890ba2878f 100644 --- a/tools/testing/selftests/hid/hid_common.h +++ b/tools/testing/selftests/hid/hid_common.h @@ -13,6 +13,7 @@ #include <linux/uhid.h> #define SHOW_UHID_DEBUG 0 +#define MAX_BUF_SIZE 128 #define min(a, b) \ ({ __typeof__(a) _a = (a); \ @@ -97,6 +98,28 @@ static unsigned char rdesc[] = { static __u8 feature_data[] = { 1, 2 }; +static __maybe_unused unsigned char fido2_rdesc[] = { + 0x06, 0xd0, 0xf1, /* Usage Page (FIDO Alliance) */ + 0x09, 0x01, /* Usage (U2F Authenticator Device) */ + 0xa1, 0x01, /* Collection (Application) */ + 0x09, 0x20, /* Usage (Input Report Data) */ + 0x15, 0x00, /* Logical Minimum (0) */ + 0x26, 0xff, 0x00, /* Logical Maximum (255) */ + 0x75, 0x08, /* Report Size (8) */ + 0x95, 0x40, /* Report Count (64) */ + 0x81, 0x02, /* Input (Data,Var,Abs) */ + 0x09, 0x21, /* Usage (Output Report Data) */ + 0x15, 0x00, /* Logical Minimum (0) */ + 0x26, 0xff, 0x00, /* Logical Maximum (255) */ + 0x75, 0x08, /* Report Size (8) */ + 0x95, 0x40, /* Report Count (64) */ + 0x91, 0x02, /* Output (Data,Var,Abs) */ + 0x06, 0x00, 0xff, /* Usage Page (Vendor Defined Page 1) */ + 0x09, 0x22, /* Usage (Vendor Usage 0x22) */ + 0xb1, 0x02, /* Feature (Data,Var,Abs) */ + 0xc0, /* End Collection */ +}; + #define ASSERT_OK(data) ASSERT_FALSE(data) #define ASSERT_OK_PTR(ptr) ASSERT_NE(NULL, ptr) @@ -110,7 +133,7 @@ static pthread_cond_t uhid_started = PTHREAD_COND_INITIALIZER; static pthread_mutex_t uhid_output_mtx = PTHREAD_MUTEX_INITIALIZER; static pthread_cond_t uhid_output_cond = PTHREAD_COND_INITIALIZER; -static unsigned char output_report[10]; +static unsigned char output_report[MAX_BUF_SIZE]; /* no need to protect uhid_stopped, only one thread accesses it */ static bool uhid_stopped; diff --git a/tools/testing/selftests/hid/progs/hid.c b/tools/testing/selftests/hid/progs/hid.c index 361dc7eaad22..48aa8088cc53 100644 --- a/tools/testing/selftests/hid/progs/hid.c +++ b/tools/testing/selftests/hid/progs/hid.c @@ -98,7 +98,7 @@ struct hid_bpf_ops change_report_id = { struct hid_hw_request_syscall_args { /* data needs to come at offset 0 so we can use it in calls */ - __u8 data[10]; + __u8 data[128]; unsigned int hid; int retval; size_t size; diff --git a/tools/testing/selftests/kvm/Makefile.kvm b/tools/testing/selftests/kvm/Makefile.kvm index 96bab7002d39..6a1482e3a286 100644 --- a/tools/testing/selftests/kvm/Makefile.kvm +++ b/tools/testing/selftests/kvm/Makefile.kvm @@ -189,6 +189,7 @@ TEST_GEN_PROGS_arm64 += arm64/stage2_block_transitions TEST_GEN_PROGS_arm64 += arm64/vcpu_width_config TEST_GEN_PROGS_arm64 += arm64/vgic_init TEST_GEN_PROGS_arm64 += arm64/vgic_irq +TEST_GEN_PROGS_arm64 += arm64/vgic_its_save TEST_GEN_PROGS_arm64 += arm64/vgic_lpi_stress TEST_GEN_PROGS_arm64 += arm64/vgic_v5 TEST_GEN_PROGS_arm64 += arm64/vpmu_counter_access diff --git a/tools/testing/selftests/kvm/arm64/smccc_filter.c b/tools/testing/selftests/kvm/arm64/smccc_filter.c index 21e41880261b..a41ed3e016ba 100644 --- a/tools/testing/selftests/kvm/arm64/smccc_filter.c +++ b/tools/testing/selftests/kvm/arm64/smccc_filter.c @@ -140,6 +140,10 @@ static void test_invalid_nr_functions(void) TEST_ASSERT(r < 0 && errno == EINVAL, "Attempt to filter 0 functions should return EINVAL"); + r = __set_smccc_filter(vm, 0, 0, KVM_SMCCC_FILTER_DENY); + TEST_ASSERT(r < 0 && errno == EINVAL, + "Attempt to filter 0 functions at base 0 should return EINVAL"); + kvm_vm_free(vm); } diff --git a/tools/testing/selftests/kvm/arm64/vgic_its_save.c b/tools/testing/selftests/kvm/arm64/vgic_its_save.c new file mode 100644 index 000000000000..864da01539f3 --- /dev/null +++ b/tools/testing/selftests/kvm/arm64/vgic_its_save.c @@ -0,0 +1,441 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * vgic_its_save - KVM_DEV_ARM_ITS_SAVE_TABLES against tables a guest broke. + * + * Both cases are reachable by a guest on its own, and neither may fail a save + * that userspace has to be able to issue: + * + * - Changing GITS_BASER<coll> drops the collections it described, so the save + * writes nothing but the terminating invalid entry. + * - A device the device table can no longer address is skipped, and the saved + * DTE chain skips it too rather than pointing at an entry never written. + * + * Both cases then reset and restore, which is what the save exists for. + * + * Copyright (c) 2026 Google LLC + * Author: Fuad Tabba <fuad.tabba@linux.dev> + */ + +#include <endian.h> +#include <linux/align.h> +#include <linux/bitfield.h> +#include <linux/sizes.h> + +#include "kvm_util.h" +#include "gic.h" +#include "gic_v3.h" +#include "gic_v3_its.h" +#include "processor.h" +#include "ucall.h" +#include "vgic.h" + +#define TEST_MEMSLOT_INDEX 1 + +/* All three ITS table entry sizes are 8 bytes in ABI 0. */ +#define ESZ 8 +#define ENTRIES_PER_PAGE (SZ_64K / ESZ) + +/* CTE and DTE layout, mirroring KVM's KVM_ITS_* in arch/arm64/kvm/vgic/vgic.h */ +#define CTE_VALID_MASK BIT_ULL(63) +#define DTE_VALID_MASK BIT_ULL(63) +#define DTE_NEXT_SHIFT 49 +#define DTE_NEXT_MASK GENMASK_ULL(62, 49) + +/* L1 entry of an indirect table: valid bit plus a 64K aligned L2 address. */ +#define L1E_VALID_MASK BIT_ULL(63) +#define L1E_ADDR_MASK GENMASK_ULL(51, 16) + +#define GITS_BASER_PAGES_MASK GENMASK_ULL(7, 0) + +#define POISON 0xdeadbeefdeadbeefULL + +/* The collection table starts at two pages and is shrunk to one. */ +#define COLL_TBL_PAGES 2 +#define COLL_TBL_SZ (COLL_TBL_PAGES * SZ_64K) + +/* One more collection than the shrunken table can hold. */ +#define NR_COLLECTIONS (ENTRIES_PER_PAGE + 1) + +/* Two devices, one per L2 block of the indirect device table. */ +#define DEVICE_A_ID 0 +#define DEVICE_B_ID ENTRIES_PER_PAGE + +/* + * its_send_mapd_cmd() encodes ilog2(itt_size) - 1 as num_eventid_bits, and + * vgic_its_restore_itt() scans BIT_ULL(num_eventid_bits) * ESZ, so the size + * handed to MAPD has to match the ITT allocated for it. + */ +#define ITT_EVENTID_BITS 13 +#define ITT_MAPD_SIZE BIT_ULL(ITT_EVENTID_BITS + 1) +#define ITT_SZ (BIT_ULL(ITT_EVENTID_BITS) * ESZ) + +static struct kvm_vm *vm; +static struct kvm_vcpu *vcpu; +static int its_fd; +static gpa_t gpa_base; + +static struct test_data { + gpa_t device_table; + gpa_t collection_table; + gpa_t cmdq_base; + void *cmdq_base_va; + + gpa_t lpi_prop_table; + gpa_t lpi_pend_table; + + void *device_l1_va; + gpa_t device_l2[2]; + gpa_t itt_tables; +} test_data; + +static unsigned long its_baser_offset(unsigned int type) +{ + int i; + + for (i = 0; i < GITS_BASER_NR_REGS; i++) { + unsigned long offset = GITS_BASER + (i * sizeof(u64)); + u64 baser = readq_relaxed(GITS_BASE_GVA + offset); + + if (GITS_BASER_TYPE(baser) == type) + return offset; + } + + GUEST_FAIL("Couldn't find an ITS BASER of type %u", type); + return -1; +} + +static void its_set_enable(bool enable) +{ + u32 ctlr = readl_relaxed(GITS_BASE_GVA + GITS_CTLR); + + if (enable) + ctlr |= GITS_CTLR_ENABLE; + else + ctlr &= ~GITS_CTLR_ENABLE; + + writel_relaxed(ctlr, GITS_BASE_GVA + GITS_CTLR); +} + +/* + * Shrink the collection table to a single page, leaving VALID set. BASER + * writes are ignored while the ITS is enabled. + */ +static void guest_shrink_coll_table(void) +{ + unsigned long offset = its_baser_offset(GITS_BASER_TYPE_COLLECTION); + u64 baser; + + its_set_enable(false); + + baser = readq_relaxed(GITS_BASE_GVA + offset); + baser &= ~GITS_BASER_PAGES_MASK; + writeq_relaxed(baser, GITS_BASE_GVA + offset); +} + +static void guest_baser_change(void) +{ + u32 coll_id; + + gic_init(GIC_V3, 1); + gic_rdist_enable_lpis(test_data.lpi_prop_table, SZ_64K, + test_data.lpi_pend_table); + + its_init(test_data.collection_table, COLL_TBL_SZ, + test_data.device_table, SZ_64K, + test_data.cmdq_base, SZ_64K); + + for (coll_id = 0; coll_id < NR_COLLECTIONS; coll_id++) + its_send_mapc_cmd(test_data.cmdq_base_va, 0, coll_id, true); + + guest_shrink_coll_table(); + + GUEST_DONE(); +} + +/* Turn the already installed device table into an indirect one. */ +static void guest_make_device_table_indirect(void) +{ + unsigned long offset = its_baser_offset(GITS_BASER_TYPE_DEVICE); + u64 baser; + + its_set_enable(false); + + baser = readq_relaxed(GITS_BASE_GVA + offset); + writeq_relaxed(baser | GITS_BASER_INDIRECT, GITS_BASE_GVA + offset); + + its_set_enable(true); +} + +static void guest_unreachable_device(void) +{ + u64 *l1; + + gic_init(GIC_V3, 1); + gic_rdist_enable_lpis(test_data.lpi_prop_table, SZ_64K, + test_data.lpi_pend_table); + + its_init(test_data.collection_table, SZ_64K, + test_data.device_table, SZ_64K, + test_data.cmdq_base, SZ_64K); + + guest_make_device_table_indirect(); + + /* Both L2 blocks present, so both MAPDs are in range. */ + l1 = test_data.device_l1_va; + l1[0] = L1E_VALID_MASK | (test_data.device_l2[0] & L1E_ADDR_MASK); + l1[1] = L1E_VALID_MASK | (test_data.device_l2[1] & L1E_ADDR_MASK); + + its_send_mapd_cmd(test_data.cmdq_base_va, DEVICE_A_ID, + test_data.itt_tables, ITT_MAPD_SIZE, true); + its_send_mapd_cmd(test_data.cmdq_base_va, DEVICE_B_ID, + test_data.itt_tables + ITT_SZ, ITT_MAPD_SIZE, true); + + /* + * Drop the block holding device B. No ITS command and no GITS_BASER + * write is involved, so nothing tells KVM the device is now + * unreachable. + */ + l1[1] = 0; + + GUEST_DONE(); +} + +static void run_guest(void) +{ + struct ucall uc; + + vcpu_run(vcpu); + switch (get_ucall(vcpu, &uc)) { + case UCALL_DONE: + break; + case UCALL_ABORT: + REPORT_GUEST_ASSERT(uc); + break; + default: + TEST_FAIL("Unexpected ucall: %lu", uc.cmd); + } +} + +static int save_tables(void) +{ + return __kvm_device_attr_set(its_fd, KVM_DEV_ARM_VGIC_GRP_CTRL, + KVM_DEV_ARM_ITS_SAVE_TABLES, NULL); +} + +static u64 its_reg_get(unsigned long offset) +{ + u64 val; + + kvm_device_attr_get(its_fd, KVM_DEV_ARM_VGIC_GRP_ITS_REGS, offset, + &val); + return val; +} + +static void its_reg_set(unsigned long offset, u64 val) +{ + kvm_device_attr_set(its_fd, KVM_DEV_ARM_VGIC_GRP_ITS_REGS, offset, + &val); +} + +/* + * What a migration target does with the saved tables, in the order + * Documentation/virt/kvm/devices/arm-vgic-its.rst gives: the GITS_ registers + * first, then the tables. The reset in between clears GITS_BASER<n>.Valid, + * which is why the registers have to be written back before the restore. + */ +static void reset_and_restore_tables(void) +{ + u64 baser[GITS_BASER_NR_REGS]; + int ret, i; + + for (i = 0; i < GITS_BASER_NR_REGS; i++) + baser[i] = its_reg_get(GITS_BASER + (i * sizeof(u64))); + + ret = __kvm_device_attr_set(its_fd, KVM_DEV_ARM_VGIC_GRP_CTRL, + KVM_DEV_ARM_ITS_CTRL_RESET, NULL); + TEST_ASSERT(!ret, "Expected the reset to succeed, got ret %d errno %d", + ret, errno); + + for (i = 0; i < GITS_BASER_NR_REGS; i++) + its_reg_set(GITS_BASER + (i * sizeof(u64)), baser[i]); + + ret = __kvm_device_attr_set(its_fd, KVM_DEV_ARM_VGIC_GRP_CTRL, + KVM_DEV_ARM_ITS_RESTORE_TABLES, NULL); + TEST_ASSERT(!ret, "Expected the restore to succeed, got ret %d errno %d", + ret, errno); +} + +static void poison_range(gpa_t base, size_t size) +{ + u64 *entry = addr_gpa2hva(vm, base); + size_t i; + + for (i = 0; i < size / ESZ; i++) + entry[i] = POISON; +} + +static void setup_memslot(size_t sz) +{ + size_t pages = sz / vm->page_size; + + gpa_base = ((vm_compute_max_gfn(vm) + 1) * vm->page_size) - sz; + vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS, gpa_base, + TEST_MEMSLOT_INDEX, pages, 0); +} + +static gpa_t alloc_64k(size_t nr) +{ + size_t pages_per_64k = vm_calc_num_guest_pages(vm->mode, SZ_64K); + gpa_t gpa = vm_phy_pages_alloc(vm, nr * pages_per_64k, gpa_base, + TEST_MEMSLOT_INDEX); + + TEST_ASSERT(IS_ALIGNED(gpa, SZ_64K), + "Allocation at 0x%lx is not 64K aligned, GITS_BASER cannot address it", + gpa); + return gpa; +} + +static void map_to_guest(gpa_t gpa, size_t nr) +{ + size_t pages_per_64k = vm_calc_num_guest_pages(vm->mode, SZ_64K); + + virt_map(vm, gpa, gpa, nr * pages_per_64k); +} + +static void setup_common(void) +{ + test_data.cmdq_base = alloc_64k(1); + map_to_guest(test_data.cmdq_base, 1); + test_data.cmdq_base_va = (void *)test_data.cmdq_base; + + test_data.lpi_prop_table = alloc_64k(1); + test_data.lpi_pend_table = alloc_64k(1); +} + +static void teardown(void) +{ + close(its_fd); + kvm_vm_free(vm); + memset(&test_data, 0, sizeof(test_data)); +} + +/* + * A GITS_BASER<coll> write that changes the table drops the collections it + * described. The save then has an empty list, so it writes the terminating + * invalid entry and nothing else. + */ +static void test_baser_change_drops_collections(void) +{ + u64 *cte; + int ret, i; + + pr_info("Testing that a GITS_BASER change drops the collections\n"); + + vm = vm_create_with_one_vcpu(&vcpu, guest_baser_change); + setup_memslot((4 + COLL_TBL_PAGES) * SZ_64K); + its_fd = vgic_its_setup(vm); + + test_data.device_table = alloc_64k(1); + test_data.collection_table = alloc_64k(COLL_TBL_PAGES); + setup_common(); + + sync_global_to_guest(vm, test_data); + run_guest(); + + /* Anything KVM writes is then the only thing that changed. */ + poison_range(test_data.collection_table, COLL_TBL_SZ); + + ret = save_tables(); + TEST_ASSERT(!ret, "Expected the save to succeed, got %d errno %d", + ret, errno); + + cte = addr_gpa2hva(vm, test_data.collection_table); + + /* + * Finding the terminator at the head of the table is also what proves + * the reads below landed in the saved table rather than elsewhere. + */ + TEST_ASSERT(le64toh(cte[0]) == 0, + "CTE 0: expected the terminating invalid entry, got 0x%llx", + (unsigned long long)le64toh(cte[0])); + + for (i = 1; i < COLL_TBL_SZ / ESZ; i++) + TEST_ASSERT(cte[i] == POISON, + "CTE %d: expected it untouched, got 0x%llx", + i, (unsigned long long)cte[i]); + + reset_and_restore_tables(); + + teardown(); +} + +/* + * A device whose L2 block the guest dropped is skipped by the save, and the + * DTE chain skips it too: left alone, the surviving device would point at an + * entry the save never wrote. + */ +static void test_unreachable_device_skipped(void) +{ + u64 dte; + int ret; + + pr_info("Testing that an unreachable device is skipped by the save\n"); + + vm = vm_create_with_one_vcpu(&vcpu, guest_unreachable_device); + setup_memslot(9 * SZ_64K); + its_fd = vgic_its_setup(vm); + + test_data.device_table = alloc_64k(1); + test_data.collection_table = alloc_64k(1); + test_data.device_l2[0] = alloc_64k(1); + test_data.device_l2[1] = alloc_64k(1); + test_data.itt_tables = alloc_64k(2); + setup_common(); + + map_to_guest(test_data.device_table, 1); + test_data.device_l1_va = (void *)test_data.device_table; + + sync_global_to_guest(vm, test_data); + run_guest(); + + poison_range(test_data.device_l2[0], SZ_64K); + poison_range(test_data.device_l2[1], SZ_64K); + + ret = save_tables(); + TEST_ASSERT(!ret, "Expected the save to succeed, got %d errno %d", + ret, errno); + + dte = le64toh(*(u64 *)addr_gpa2hva(vm, test_data.device_l2[0])); + + /* Device A is still reachable, so it is saved. */ + TEST_ASSERT(dte & DTE_VALID_MASK, + "Device A: expected a valid DTE, got 0x%llx", + (unsigned long long)dte); + + /* + * Device B is the only device after it and was skipped, so nothing + * follows A in the saved chain. + */ + TEST_ASSERT(FIELD_GET(DTE_NEXT_MASK, dte) == 0, + "Device A: expected no next device, got offset %llu", + (unsigned long long)FIELD_GET(DTE_NEXT_MASK, dte)); + + /* And nothing was written into the block the guest dropped. */ + TEST_ASSERT(*(u64 *)addr_gpa2hva(vm, test_data.device_l2[1]) == POISON, + "Device B: expected its entry untouched"); + + reset_and_restore_tables(); + + teardown(); +} + +int main(void) +{ + TEST_REQUIRE(kvm_supports_vgic_v3()); + + test_baser_change_drops_collections(); + test_unreachable_device_skipped(); + + pr_info("All ok!\n"); + return 0; +} diff --git a/tools/testing/selftests/kvm/hardware_disable_test.c b/tools/testing/selftests/kvm/hardware_disable_test.c index 43a36ef3ead8..1c20892d6782 100644 --- a/tools/testing/selftests/kvm/hardware_disable_test.c +++ b/tools/testing/selftests/kvm/hardware_disable_test.c @@ -37,7 +37,7 @@ static void *run_vcpu(void *arg) struct kvm_vcpu *vcpu = arg; struct kvm_run *run = vcpu->run; -#ifndef _GNU_SOURCE +#ifndef __GLIBC__ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set); #endif @@ -51,7 +51,7 @@ static void *sleeping_thread(void *arg) { int fd; -#ifndef _GNU_SOURCE +#ifndef __GLIBC__ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set); #endif @@ -71,7 +71,7 @@ static void run_test(u32 run) u32 i, j; TEST_ASSERT_EQ(pthread_attr_init(&attr), 0); -#ifdef _GNU_SOURCE +#ifdef __GLIBC__ TEST_ASSERT_EQ(pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &threads_cpu_set), 0); #endif diff --git a/tools/testing/selftests/kvm/steal_time.c b/tools/testing/selftests/kvm/steal_time.c index bc3c62b72c58..785d19f9ee8f 100644 --- a/tools/testing/selftests/kvm/steal_time.c +++ b/tools/testing/selftests/kvm/steal_time.c @@ -27,6 +27,9 @@ static void *st_gva[NR_VCPUS]; static u64 guest_stolen_time[NR_VCPUS]; +static struct kvm_vm *vm_create_steal_time(u32 nr_vcpus, void *guest_code, + struct kvm_vcpu *vcpus[]); + #if defined(__x86_64__) /* steal_time must have 64-byte alignment */ @@ -210,17 +213,14 @@ static void check_steal_time_uapi(void) u64 st_ipa; int ret; - vm = vm_create_with_one_vcpu(&vcpu, NULL); - struct kvm_device_attr dev = { .group = KVM_ARM_VCPU_PVTIME_CTRL, .attr = KVM_ARM_VCPU_PVTIME_IPA, .addr = (u64)&st_ipa, }; + vm = vm_create_steal_time(1, NULL, &vcpu); vcpu_ioctl(vcpu, KVM_HAS_DEVICE_ATTR, &dev); - vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS, ST_GPA_BASE, 1, 1, 0); - virt_map(vm, ST_GPA_BASE, ST_GPA_BASE, 1); st_ipa = (ulong)ST_GPA_BASE | 1; ret = __vcpu_ioctl(vcpu, KVM_SET_DEVICE_ATTR, &dev); @@ -500,13 +500,27 @@ static void run_vcpu(struct kvm_vcpu *vcpu) } } +static struct kvm_vm *vm_create_steal_time(u32 nr_vcpus, void *guest_code, + struct kvm_vcpu *vcpus[]) +{ + unsigned int gpages; + struct kvm_vm *vm; + + /* Create a VM and an identity mapped memslot for the steal time structure */ + vm = vm_create_with_vcpus(nr_vcpus, guest_code, vcpus); + gpages = vm_calc_num_guest_pages(VM_MODE_DEFAULT, STEAL_TIME_SIZE * nr_vcpus); + vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS, ST_GPA_BASE, 1, gpages, 0); + virt_map(vm, ST_GPA_BASE, ST_GPA_BASE, gpages); + + return vm; +} + int main(int ac, char **av) { struct kvm_vcpu *vcpus[NR_VCPUS]; struct kvm_vm *vm; pthread_t thread; cpu_set_t cpuset; - unsigned int gpages; long stolen_time; long run_delay; bool verbose; @@ -517,11 +531,7 @@ int main(int ac, char **av) /* Set CPU affinity so we can force preemption of the VCPU */ cpu = pin_self_to_any_cpu(); - /* Create a VM and an identity mapped memslot for the steal time structure */ - vm = vm_create_with_vcpus(NR_VCPUS, guest_code, vcpus); - gpages = vm_calc_num_guest_pages(VM_MODE_DEFAULT, STEAL_TIME_SIZE * NR_VCPUS); - vm_userspace_mem_region_add(vm, VM_MEM_SRC_ANONYMOUS, ST_GPA_BASE, 1, gpages, 0); - virt_map(vm, ST_GPA_BASE, ST_GPA_BASE, gpages); + vm = vm_create_steal_time(NR_VCPUS, guest_code, vcpus); ksft_print_header(); TEST_REQUIRE(is_steal_time_supported(vcpus[0])); diff --git a/tools/testing/selftests/landlock/fs_test.c b/tools/testing/selftests/landlock/fs_test.c index 18dbdb99aeba..6e979cef884d 100644 --- a/tools/testing/selftests/landlock/fs_test.c +++ b/tools/testing/selftests/landlock/fs_test.c @@ -10493,9 +10493,9 @@ FIXTURE_TEARDOWN_PARENT(trace_layout1) } /* - * Verifies that check_rule_fs events include correct field values: domain, dev, - * ino, access_request, and grants. All values are verified against stat() of - * the rule path on a deterministic tmpfs layout. + * Verifies that check_rule_inode events include correct field values: domain, + * dev, ino, access_request, and grants. All values are verified against stat() + * of the rule path on a deterministic tmpfs layout. */ TEST_F(trace_layout1, check_rule_fs_fields) { @@ -10529,7 +10529,7 @@ TEST_F(trace_layout1, check_rule_fs_fields) EXPECT_EQ(1, tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK))) { - TH_LOG("Expected 1 check_rule_fs event\n%s", buf); + TH_LOG("Expected 1 check_rule_inode event\n%s", buf); } ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK), @@ -10570,8 +10570,8 @@ TEST_F(trace_layout1, check_rule_fs_fields) } /* - * Verifies check_rule_fs behavior with multiple rules. With rules at s1d1 and - * s1d2 (a child of s1d1), accessing s1d2 produces only 1 event because the + * Verifies check_rule_inode behavior with multiple rules. With rules at s1d1 + * and s1d2 (a child of s1d1), accessing s1d2 produces only 1 event because the * pathwalk short-circuits after the first rule fully unmasks the single layer. */ TEST_F(trace_layout1, check_rule_fs_multiple_rules) @@ -10643,14 +10643,14 @@ TEST_F(trace_layout1, check_rule_fs_multiple_rules) ASSERT_NE(NULL, buf); /* - * Only 1 check_rule_fs event: the rule on dir_s1d2 fully unmasked the - * single layer, so the pathwalk short-circuits before reaching the + * Only one check_rule_inode event: the rule on dir_s1d2 fully unmasks + * the single layer, so the pathwalk short-circuits before reaching the * dir_s1d1 rule. */ count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK)); EXPECT_EQ(1, count) { - TH_LOG("Expected 1 check_rule_fs event, got %d\n%s", count, + TH_LOG("Expected 1 check_rule_inode event, got %d\n%s", count, buf); } @@ -10777,7 +10777,7 @@ TEST_F(trace_layout1, check_rule_fs_optional_access) count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK)); EXPECT_EQ(1, count) { - TH_LOG("Expected 1 check_rule_fs event, got %d\n%s", count, + TH_LOG("Expected 1 check_rule_inode event, got %d\n%s", count, buf); } @@ -10796,7 +10796,7 @@ TEST_F(trace_layout1, check_rule_fs_optional_access) } /* - * Verifies that check_rule_fs fires for a rule that matches the inode even when + * Verifies that check_rule_inode fires for a rule matching the inode even when * it grants none of the requested rights, so the grants set is empty. Landlock * cannot know a rule ignores the request before reading it, so the event is * still emitted (grants={}), which lets a tracer see that the rule matched. @@ -10884,7 +10884,7 @@ TEST_F(trace_layout1, check_rule_fs_empty_grant) count = tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK)); EXPECT_EQ(2, count) { - TH_LOG("Expected 2 check_rule_fs events, got %d\n%s", count, + TH_LOG("Expected 2 check_rule_inode events, got %d\n%s", count, buf); } @@ -10894,7 +10894,7 @@ TEST_F(trace_layout1, check_rule_fs_empty_grant) tracefs_count_matches( buf, TRACE_PREFIX( - TRACE_TASK) "landlock_check_rule_fs: domain=[0-9a-f]\\+ " + TRACE_TASK) "landlock_check_rule_inode: domain=[0-9a-f]\\+ " "access_request=read_dir " "dev=[0-9]\\+:[0-9]\\+ ino=[0-9]\\+ " "grants={}$")) @@ -10908,7 +10908,7 @@ TEST_F(trace_layout1, check_rule_fs_empty_grant) tracefs_count_matches( buf, TRACE_PREFIX( - TRACE_TASK) "landlock_check_rule_fs: domain=[0-9a-f]\\+ " + TRACE_TASK) "landlock_check_rule_inode: domain=[0-9a-f]\\+ " "access_request=read_dir " "dev=[0-9]\\+:[0-9]\\+ ino=[0-9]\\+ " "grants={read_dir}$")) diff --git a/tools/testing/selftests/landlock/net_test.c b/tools/testing/selftests/landlock/net_test.c index a18761e0fd82..16afbfdf06bb 100644 --- a/tools/testing/selftests/landlock/net_test.c +++ b/tools/testing/selftests/landlock/net_test.c @@ -3481,13 +3481,14 @@ TEST_F(trace_net, deny_access_net_bind) } /* - * Anchors the denial fields shared by every deny_access_net event so a field - * test proves more than sport/dport: the denying domain, the same-exec bit, the - * audit-logging verdict, and the blocked access all stay populated. + * Anchors the denial fields shared by every deny_access_net event so a port + * test also proves the denying domain, execution status, logging verdict, and + * exact blocked access. */ static void expect_net_deny_common_fields(struct __test_metadata *const _metadata, - const char *const buf) + const char *const buf, + const char *const expected_blockers) { char field[64]; @@ -3511,18 +3512,21 @@ expect_net_deny_common_fields(struct __test_metadata *const _metadata, ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK), "blockers", field, sizeof(field))); - EXPECT_STRNE("", field); + EXPECT_STREQ(expected_blockers, field); } -/* Connect and field-check tests use a separate fixture without variants. */ +enum trace_net_operation { + TRACE_NET_BIND, + TRACE_NET_SEND, +}; /* clang-format off */ -FIXTURE(trace_net_connect) { +FIXTURE(trace_net_address) { /* clang-format on */ int tracefs_ok; }; -FIXTURE_SETUP(trace_net_connect) +FIXTURE_SETUP(trace_net_address) { int ret; @@ -3547,7 +3551,7 @@ FIXTURE_SETUP(trace_net_connect) clear_cap(_metadata, CAP_SYS_ADMIN); } -FIXTURE_TEARDOWN(trace_net_connect) +FIXTURE_TEARDOWN(trace_net_address) { if (!self->tracefs_ok) return; @@ -3559,160 +3563,183 @@ FIXTURE_TEARDOWN(trace_net_connect) } /* clang-format off */ -FIXTURE_VARIANT(trace_net_connect) { +FIXTURE_VARIANT(trace_net_address) { /* clang-format on */ - /* handled_access_net, also the access allowed on the base port. */ - __u64 handled; - /* Bind the allowed base port before the denied operation. */ - bool bind_base_first; - /* Denied operation on the next port: connect (true) or bind (false). */ - bool deny_connect; + int socket_family; + int socket_type; + enum trace_net_operation operation; + int address_family; + socklen_t addrlen; + __u64 handled_access; + const char *expected_blockers; + bool address_port_zero; + bool expected_address_port; + int expected_port; }; /* clang-format off */ - -/* Denied connect(): sport=0, dport=<denied port>. */ -FIXTURE_VARIANT_ADD(trace_net_connect, connect_denied) { - .handled = LANDLOCK_ACCESS_NET_CONNECT_TCP, - .bind_base_first = false, - .deny_connect = true, +FIXTURE_VARIANT_ADD(trace_net_address, ipv4_tcp_bind) { + /* clang-format on */ + .socket_family = AF_INET, + .socket_type = SOCK_STREAM, + .operation = TRACE_NET_BIND, + .address_family = AF_INET, + .addrlen = sizeof(struct sockaddr_in), + .handled_access = LANDLOCK_ACCESS_NET_BIND_TCP, + .expected_blockers = "bind_tcp", + .expected_address_port = true, }; -/* Denied bind(): sport=<denied port>, dport=0. */ -FIXTURE_VARIANT_ADD(trace_net_connect, bind_fields) { - .handled = LANDLOCK_ACCESS_NET_BIND_TCP, - .bind_base_first = false, - .deny_connect = false, +/* Explicit bind(0) has a checked zero port. */ +/* clang-format off */ +FIXTURE_VARIANT_ADD(trace_net_address, ipv4_udp_bind_zero) { + /* clang-format on */ + .socket_family = AF_INET, + .socket_type = SOCK_DGRAM, + .operation = TRACE_NET_BIND, + .address_family = AF_INET, + .addrlen = sizeof(struct sockaddr_in), + .handled_access = LANDLOCK_ACCESS_NET_BIND_UDP, + .expected_blockers = "bind_udp", + .address_port_zero = true, + .expected_port = 0, +}; + +/* A UDP send can deny its synthetic unspecified bind endpoint. */ +/* clang-format off */ +FIXTURE_VARIANT_ADD(trace_net_address, ipv6_udp_autobind) { + /* clang-format on */ + .socket_family = AF_INET6, + .socket_type = SOCK_DGRAM, + .operation = TRACE_NET_SEND, + .address_family = AF_INET6, + .addrlen = sizeof(struct sockaddr_in6), + .handled_access = LANDLOCK_ACCESS_NET_BIND_UDP, + .expected_blockers = "bind_udp", + .expected_port = 0, }; -/* Denied connect() after an allowed bind(): the connect fields (sport=0). */ -FIXTURE_VARIANT_ADD(trace_net_connect, connect_after_bind) { - .handled = LANDLOCK_ACCESS_NET_BIND_TCP | LANDLOCK_ACCESS_NET_CONNECT_TCP, - .bind_base_first = true, - .deny_connect = true, +/* A family-only address has no checked port. */ +/* clang-format off */ +FIXTURE_VARIANT_ADD(trace_net_address, ipv6_unspec_udp_send_min) { + /* clang-format on */ + .socket_family = AF_INET6, + .socket_type = SOCK_DGRAM, + .operation = TRACE_NET_SEND, + .address_family = AF_UNSPEC, + .addrlen = sizeof(sa_family_t), + .handled_access = LANDLOCK_ACCESS_NET_CONNECT_SEND_UDP, + .expected_blockers = "connect_send_udp", + .expected_port = -1, }; -/* clang-format on */ +static void set_trace_net_address(struct sockaddr_storage *const storage, + const int socket_family, + const int address_family, + const unsigned short port) +{ + memset(storage, 0, sizeof(*storage)); -/* - * A denied TCP bind(2) or connect(2) emits one deny_access_net event. The port - * is reported in the field matching the denied operation, in host endianness - * (the UAPI landlock_net_port_attr.port convention): a connect denial reports - * sport=0 dport=<port>, a bind denial reports sport=<port> dport=0, so a - * byte-order or field-swap bug is caught. A prior allowed bind - * (connect_after_bind) does not change the connect denial's fields. - */ -TEST_F(trace_net_connect, deny_access_net) + if (socket_family == AF_INET) { + struct sockaddr_in *const addr4 = (struct sockaddr_in *)storage; + + addr4->sin_family = address_family; + addr4->sin_port = htons(port); + addr4->sin_addr.s_addr = htonl(INADDR_LOOPBACK); + } else { + struct sockaddr_in6 *const addr6 = + (struct sockaddr_in6 *)storage; + + addr6->sin6_family = address_family; + addr6->sin6_port = htons(port); + addr6->sin6_addr = in6addr_loopback; + } +} + +/* Verifies the actionable signed port for representative checked shapes. */ +TEST_F(trace_net_address, deny_access_net) { - pid_t child; - int status; - char *buf; + const char *const event_regex = REGEX_DENY_ACCESS_NET(TRACE_TASK); + const unsigned short address_port = + variant->address_port_zero ? 0 : sock_port_start + 1; + const int expected_port = variant->expected_address_port ? + address_port : + variant->expected_port; + const struct landlock_ruleset_attr ruleset_attr = { + .handled_access_net = variant->handled_access, + }; + struct sockaddr_storage address; char field[64], expected[16]; + char *buf; + int count, ret, ruleset_fd, socket_fd, status; + pid_t child; if (!self->tracefs_ok) SKIP(return, "tracefs not available"); + set_trace_net_address(&address, variant->socket_family, + variant->address_family, address_port); + socket_fd = socket(variant->socket_family, + variant->socket_type | SOCK_CLOEXEC, 0); + ASSERT_LE(0, socket_fd); + ruleset_fd = + landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0); + ASSERT_LE(0, ruleset_fd); + ASSERT_EQ(0, tracefs_clear_buf()); + child = fork(); ASSERT_LE(0, child); - if (child == 0) { - struct landlock_ruleset_attr ruleset_attr = { - .handled_access_net = variant->handled, - }; - struct landlock_net_port_attr port_attr = { - .allowed_access = variant->handled, - .port = sock_port_start, - }; - struct sockaddr_in addr = { - .sin_family = AF_INET, - .sin_addr.s_addr = htonl(INADDR_LOOPBACK), - }; - int ruleset_fd, sock_fd, optval = 1, ret; - - ruleset_fd = landlock_create_ruleset(&ruleset_attr, - sizeof(ruleset_attr), 0); - if (ruleset_fd < 0) + if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0)) _exit(1); - if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NET_PORT, - &port_attr, 0)) { - close(ruleset_fd); - _exit(1); - } - prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0); - if (landlock_restrict_self(ruleset_fd, 0)) { - close(ruleset_fd); - _exit(1); - } + if (landlock_restrict_self(ruleset_fd, 0)) + _exit(2); close(ruleset_fd); - sock_fd = socket(AF_INET, SOCK_STREAM | SOCK_CLOEXEC, 0); - if (sock_fd < 0) - _exit(1); - - /* Bind the allowed base port first (succeeds, no event). */ - if (variant->bind_base_first) { - setsockopt(sock_fd, SOL_SOCKET, SO_REUSEADDR, &optval, - sizeof(optval)); - addr.sin_port = htons(sock_port_start); - if (bind(sock_fd, (struct sockaddr *)&addr, - sizeof(addr))) { - close(sock_fd); - _exit(1); - } - } - - /* Denied operation on the next port. */ - addr.sin_port = htons(sock_port_start + 1); - if (variant->deny_connect) - ret = connect(sock_fd, (struct sockaddr *)&addr, - sizeof(addr)); - else - ret = bind(sock_fd, (struct sockaddr *)&addr, - sizeof(addr)); - if (ret == 0) { - close(sock_fd); - _exit(2); - } - if (errno != EACCES) { - close(sock_fd); + switch (variant->operation) { + case TRACE_NET_BIND: + ret = bind(socket_fd, (const struct sockaddr *)&address, + variant->addrlen); + break; + case TRACE_NET_SEND: + ret = sendto(socket_fd, "A", 1, MSG_NOSIGNAL, + (const struct sockaddr *)&address, + variant->addrlen); + break; + default: _exit(3); } - close(sock_fd); + if (ret >= 0 || errno != EACCES) + _exit(4); + close(socket_fd); + _exit(0); } + close(ruleset_fd); + close(socket_fd); ASSERT_EQ(child, waitpid(child, &status, 0)); ASSERT_TRUE(WIFEXITED(status)); - EXPECT_EQ(0, WEXITSTATUS(status)); + ASSERT_EQ(0, WEXITSTATUS(status)); buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - - EXPECT_EQ(1, tracefs_count_matches(buf, - REGEX_DENY_ACCESS_NET(TRACE_TASK))); - - expect_net_deny_common_fields(_metadata, buf); - - /* - * The denied operation's port field carries the port; the other is 0. - */ - snprintf(expected, sizeof(expected), "%llu", - (unsigned long long)(sock_port_start + 1)); - - ASSERT_EQ(0, - tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK), - "sport", field, sizeof(field))); - EXPECT_STREQ(variant->deny_connect ? "0" : expected, field); - - ASSERT_EQ(0, - tracefs_extract_field(buf, REGEX_DENY_ACCESS_NET(TRACE_TASK), - "dport", field, sizeof(field))); - EXPECT_STREQ(variant->deny_connect ? expected : "0", field); + count = tracefs_count_matches(buf, event_regex); + if (count != 1) + TH_LOG("Expected 1 denial event, got %d\n%s", count, buf); + ASSERT_EQ(1, count); + expect_net_deny_common_fields(_metadata, buf, + variant->expected_blockers); + + ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "port", field, + sizeof(field))); + snprintf(expected, sizeof(expected), "%d", expected_port); + EXPECT_STREQ(expected, field); free(buf); } -/* Field verification for the check_rule_net event on an allowed access. */ +/* Field verification for the check_rule_net_port event on an allowed access. */ /* clang-format off */ FIXTURE(trace_net_check_rule) { @@ -3757,10 +3784,11 @@ FIXTURE_TEARDOWN(trace_net_check_rule) /* * Verifies that an allowed bind matching a net-port rule emits exactly one - * landlock_check_rule_net event with the enforcing domain, the requested + * landlock_check_rule_net_port event with the enforcing domain, the requested * access, the checked port (host endianness), and the per-layer grants. The - * whole event is anchored to exact values so a revert of the check_rule_net - * emit (or a byte-order or field-plumbing regression) fails the test. + * whole event is anchored to exact values so removing the check_rule_net_port + * emission or introducing a byte-order or field-plumbing regression fails the + * test. */ TEST_F(trace_net_check_rule, check_rule_net_fields) { @@ -3832,7 +3860,7 @@ TEST_F(trace_net_check_rule, check_rule_net_fields) EXPECT_EQ(1, tracefs_count_matches(buf, REGEX_CHECK_RULE_NET(TRACE_TASK))) { - TH_LOG("Expected 1 check_rule_net event\n%s", buf); + TH_LOG("Expected 1 check_rule_net_port event\n%s", buf); } ASSERT_EQ(0, @@ -3866,11 +3894,4 @@ TEST_F(trace_net_check_rule, check_rule_net_fields) free(buf); } -/* - * IPv6 network trace tests are intentionally elided. IPv6 hook dispatch uses - * the same current_check_access_socket() code path as IPv4, validated by the - * audit tests in this file. The trace events use the same blockers/sport/dport - * fields regardless of address family. - */ - TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h index ba0c5e92001f..2ec863362173 100644 --- a/tools/testing/selftests/landlock/trace.h +++ b/tools/testing/selftests/landlock/trace.h @@ -27,14 +27,14 @@ TRACEFS_LANDLOCK_DIR "/landlock_create_domain/enable" #define TRACEFS_ENFORCE_DOMAIN_ENABLE \ TRACEFS_LANDLOCK_DIR "/landlock_enforce_domain/enable" -#define TRACEFS_ADD_RULE_FS_ENABLE \ - TRACEFS_LANDLOCK_DIR "/landlock_add_rule_fs/enable" -#define TRACEFS_ADD_RULE_NET_ENABLE \ - TRACEFS_LANDLOCK_DIR "/landlock_add_rule_net/enable" +#define TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE \ + TRACEFS_LANDLOCK_DIR "/landlock_add_rule_path_beneath/enable" +#define TRACEFS_ADD_RULE_NET_PORT_ENABLE \ + TRACEFS_LANDLOCK_DIR "/landlock_add_rule_net_port/enable" #define TRACEFS_CHECK_RULE_FS_ENABLE \ - TRACEFS_LANDLOCK_DIR "/landlock_check_rule_fs/enable" + TRACEFS_LANDLOCK_DIR "/landlock_check_rule_inode/enable" #define TRACEFS_CHECK_RULE_NET_ENABLE \ - TRACEFS_LANDLOCK_DIR "/landlock_check_rule_net/enable" + TRACEFS_LANDLOCK_DIR "/landlock_check_rule_net_port/enable" #define TRACEFS_DENY_ACCESS_FS_ENABLE \ TRACEFS_LANDLOCK_DIR "/landlock_deny_access_fs/enable" #define TRACEFS_DENY_ACCESS_NET_ENABLE \ @@ -79,18 +79,18 @@ */ #define KWORKER_TASK "kworker/[0-9]\\+:[0-9]\\+" -#define REGEX_ADD_RULE_FS(task) \ - TRACE_PREFIX(task) \ - "landlock_add_rule_fs: " \ - "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \ - "access_rights=[a-z_|]* " \ - "dev=[0-9]\\+:[0-9]\\+ " \ - "ino=[0-9]\\+ " \ +#define REGEX_ADD_RULE_PATH_BENEATH(task) \ + TRACE_PREFIX(task) \ + "landlock_add_rule_path_beneath: " \ + "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \ + "access_rights=[a-z_|]* " \ + "dev=[0-9]\\+:[0-9]\\+ " \ + "ino=[0-9]\\+ " \ "path=[^ ]\\+$" -#define REGEX_ADD_RULE_NET(task) \ +#define REGEX_ADD_RULE_NET_PORT(task) \ TRACE_PREFIX(task) \ - "landlock_add_rule_net: " \ + "landlock_add_rule_net_port: " \ "ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \ "access_rights=[a-z_|]* " \ "port=[0-9]\\+$" @@ -110,21 +110,21 @@ "parent=[0-9a-f]\\+ " \ "ruleset=[0-9a-f]\\+\\.[0-9]\\+$" -#define REGEX_CHECK_RULE_FS(task) \ - TRACE_PREFIX(task) \ - "landlock_check_rule_fs: " \ - "domain=[0-9a-f]\\+ " \ - "access_request=[a-z_|]* " \ - "dev=[0-9]\\+:[0-9]\\+ " \ - "ino=[0-9]\\+ " \ +#define REGEX_CHECK_RULE_FS(task) \ + TRACE_PREFIX(task) \ + "landlock_check_rule_inode: " \ + "domain=[0-9a-f]\\+ " \ + "access_request=[a-z_|]* " \ + "dev=[0-9]\\+:[0-9]\\+ " \ + "ino=[0-9]\\+ " \ "grants={[a-z_|,]*}$" -#define REGEX_CHECK_RULE_NET(task) \ - TRACE_PREFIX(task) \ - "landlock_check_rule_net: " \ - "domain=[0-9a-f]\\+ " \ - "access_request=[a-z_|]* " \ - "port=[0-9]\\+ " \ +#define REGEX_CHECK_RULE_NET(task) \ + TRACE_PREFIX(task) \ + "landlock_check_rule_net_port: " \ + "domain=[0-9a-f]\\+ " \ + "access_request=[a-z_|]* " \ + "port=[0-9]\\+ " \ "grants={[a-z_|,]*}$" #define REGEX_DENY_ACCESS_FS(task) \ @@ -145,8 +145,7 @@ "same_exec=[01] " \ "logged=[01] " \ "blockers=[a-z_|]* " \ - "sport=[0-9]\\+ " \ - "dport=[0-9]\\+$" + "port=-\\?[0-9]\\+$" #define REGEX_DENY_PTRACE(task) \ TRACE_PREFIX(task) \ diff --git a/tools/testing/selftests/landlock/trace_fs_test.c b/tools/testing/selftests/landlock/trace_fs_test.c index 4543a25c1f55..64014ade3a0e 100644 --- a/tools/testing/selftests/landlock/trace_fs_test.c +++ b/tools/testing/selftests/landlock/trace_fs_test.c @@ -121,7 +121,8 @@ FIXTURE_SETUP(trace_fs) } self->tracefs_ok = 1; - ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, true)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, + true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, true)); ASSERT_EQ(0, tracefs_clear()); @@ -134,7 +135,7 @@ FIXTURE_TEARDOWN(trace_fs) return; set_cap(_metadata, CAP_SYS_ADMIN); - tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, false); + tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, false); tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false); tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, false); tracefs_fixture_teardown(); @@ -183,11 +184,11 @@ TEST_F(trace_fs, unsandboxed) } /* - * Verifies that adding a filesystem rule emits a landlock_add_rule_fs trace - * event with the expected path and field values: ruleset ID is non-zero, - * access_rights is non-zero, and path matches. + * Verifies that adding a filesystem rule emits a landlock_add_rule_path_beneath + * event with the expected path and field values: the ruleset ID and + * access_rights are non-zero, and the path matches. */ -TEST_F(trace_fs, add_rule_fs) +TEST_F(trace_fs, add_rule_path_beneath) { struct landlock_ruleset_attr ruleset_attr = { .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE | @@ -215,28 +216,30 @@ TEST_F(trace_fs, add_rule_fs) buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - count = tracefs_count_matches(buf, REGEX_ADD_RULE_FS(TRACE_TASK)); + count = tracefs_count_matches(buf, + REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK)); EXPECT_EQ(1, count) { - TH_LOG("Expected 1 add_rule_fs event, got %d\n%s", count, buf); + TH_LOG("Expected 1 add_rule_path_beneath event, got %d\n%s", + count, buf); } /* Ruleset ID should be non-zero. */ - ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_ADD_RULE_FS(TRACE_TASK), - "ruleset", field_buf, - sizeof(field_buf))); + ASSERT_EQ(0, tracefs_extract_field( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK), + "ruleset", field_buf, sizeof(field_buf))); EXPECT_STRNE("0", field_buf); /* Access rights should be non-zero. */ - ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_ADD_RULE_FS(TRACE_TASK), - "access_rights", field_buf, - sizeof(field_buf))); + ASSERT_EQ(0, tracefs_extract_field( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK), + "access_rights", field_buf, sizeof(field_buf))); EXPECT_STRNE("", field_buf); /* Path should be /usr. */ - ASSERT_EQ(0, - tracefs_extract_field(buf, REGEX_ADD_RULE_FS(TRACE_TASK), - "path", field_buf, sizeof(field_buf))); + ASSERT_EQ(0, tracefs_extract_field( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK), + "path", field_buf, sizeof(field_buf))); EXPECT_STREQ("/usr", field_buf); free(buf); @@ -246,7 +249,7 @@ TEST_F(trace_fs, add_rule_fs) * Verifies that a path whose escaping exceeds the trace scratch sequence does * not corrupt a sibling symbolic field. */ -TEST_F(trace_fs, add_rule_fs_escaped_path_overflow) +TEST_F(trace_fs, add_rule_path_beneath_escaped_path_overflow) { static const char access_prefix[] = "execute|write_file|read_file|"; static const char access_suffix[] = "|ioctl_dev|resolve_unix"; @@ -277,10 +280,12 @@ TEST_F(trace_fs, add_rule_fs_escaped_path_overflow) buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - count = tracefs_count_matches(buf, REGEX_ADD_RULE_FS(TRACE_TASK)); + count = tracefs_count_matches(buf, + REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK)); EXPECT_EQ(1, count) { - TH_LOG("Expected 1 add_rule_fs event, got %d\n%s", count, buf); + TH_LOG("Expected 1 add_rule_path_beneath event, got %d\n%s", + count, buf); } /* @@ -288,9 +293,9 @@ TEST_F(trace_fs, add_rule_fs_escaped_path_overflow) * field also catches scratch-sequence poisoning when the compiler * evaluates the overflowing path first, as GCC currently does. */ - ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_ADD_RULE_FS(TRACE_TASK), - "access_rights", field_buf, - sizeof(field_buf))); + ASSERT_EQ(0, tracefs_extract_field( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK), + "access_rights", field_buf, sizeof(field_buf))); EXPECT_EQ(0, strncmp(field_buf, access_prefix, sizeof(access_prefix) - 1)); EXPECT_EQ(NULL, strstr(field_buf, "|refer|")); @@ -298,7 +303,8 @@ TEST_F(trace_fs, add_rule_fs_escaped_path_overflow) ASSERT_LE(sizeof(access_suffix) - 1, field_len); EXPECT_STREQ(access_suffix, field_buf + field_len - (sizeof(access_suffix) - 1)); - expect_truncated_path(_metadata, buf, REGEX_ADD_RULE_FS(TRACE_TASK)); + expect_truncated_path(_metadata, buf, + REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK)); free(buf); } @@ -542,7 +548,8 @@ TEST_F(trace_fs, check_rule_nested) */ TEST_F(trace_fs, deny_access_fs_denied) { - char *buf; + const char *const event_regex = REGEX_DENY_ACCESS_FS(TRACE_TASK); + char *buf, blockers[64]; int count; ASSERT_EQ(0, tracefs_clear_buf()); @@ -558,8 +565,77 @@ TEST_F(trace_fs, deny_access_fs_denied) buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - count = tracefs_count_matches(buf, REGEX_DENY_ACCESS_FS(TRACE_TASK)); - EXPECT_LE(1, count); + count = tracefs_count_matches(buf, event_regex); + EXPECT_EQ(1, count) + { + TH_LOG("Expected 1 access denial, got %d\n%s", count, buf); + } + ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "blockers", + blockers, sizeof(blockers))); + EXPECT_STREQ("read_dir", blockers); + + free(buf); +} + +/* + * Verifies that a denied mount reports the singleton topology blocker rather + * than an empty access mask. + */ +TEST_F(trace_fs, deny_change_topology) +{ + const char *const event_regex = REGEX_DENY_ACCESS_FS(TRACE_TASK); + const struct landlock_ruleset_attr ruleset_attr = { + .handled_access_fs = LANDLOCK_ACCESS_FS_REFER, + }; + char *buf, blockers[64]; + int count, ruleset_fd, status; + pid_t pid; + + ruleset_fd = + landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0); + ASSERT_LE(0, ruleset_fd); + ASSERT_EQ(0, tracefs_clear_buf()); + + /* Ensure that Landlock is the only expected mount denial. */ + set_cap(_metadata, CAP_SYS_ADMIN); + pid = fork(); + ASSERT_LE(0, pid); + if (pid == 0) { + if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0)) { + close(ruleset_fd); + _exit(1); + } + if (landlock_restrict_self(ruleset_fd, 0)) { + close(ruleset_fd); + _exit(2); + } + close(ruleset_fd); + + if (mount(NULL, "/", NULL, MS_PRIVATE | MS_REC, NULL) != -1) + _exit(3); + + if (errno != EPERM) + _exit(4); + + _exit(0); + } + close(ruleset_fd); + clear_cap(_metadata, CAP_SYS_ADMIN); + + ASSERT_EQ(pid, waitpid(pid, &status, 0)); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + buf = tracefs_read_buf(); + ASSERT_NE(NULL, buf); + count = tracefs_count_matches(buf, event_regex); + EXPECT_EQ(1, count) + { + TH_LOG("Expected 1 topology denial, got %d\n%s", count, buf); + } + ASSERT_EQ(0, tracefs_extract_field(buf, event_regex, "blockers", + blockers, sizeof(blockers))); + EXPECT_STREQ("change_topology", blockers); free(buf); } diff --git a/tools/testing/selftests/landlock/trace_test.c b/tools/testing/selftests/landlock/trace_test.c index afdaf8511b3a..f9b293a9dd56 100644 --- a/tools/testing/selftests/landlock/trace_test.c +++ b/tools/testing/selftests/landlock/trace_test.c @@ -49,8 +49,10 @@ FIXTURE_SETUP(trace) ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, true)); - ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, true)); - ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, true)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, + true)); + ASSERT_EQ(0, + tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, true)); @@ -72,8 +74,8 @@ FIXTURE_TEARDOWN(trace) tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false); tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false); tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false); - tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, false); - tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, false); + tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, false); + tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE, false); tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false); tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, false); tracefs_enable_event(TRACEFS_DENY_ACCESS_FS_ENABLE, false); @@ -103,8 +105,10 @@ TEST_F(trace, no_trace_when_disabled) ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false)); - ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, false)); - ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, false)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_PATH_BENEATH_ENABLE, + false)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_PORT_ENABLE, + false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_NET_ENABLE, false)); @@ -265,10 +269,11 @@ TEST_F(trace, ruleset_version) ASSERT_NE(0, !!dot); EXPECT_STREQ("0", dot + 1); - /* Verify 2 add_rule_fs events were emitted. */ - EXPECT_EQ(2, tracefs_count_matches(buf, REGEX_ADD_RULE_FS(TRACE_TASK))) + /* Verify two add_rule_path_beneath events were emitted. */ + EXPECT_EQ(2, tracefs_count_matches( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK))) { - TH_LOG("Expected 2 add_rule_fs events\n%s", buf); + TH_LOG("Expected 2 add_rule_path_beneath events\n%s", buf); } /* @@ -373,7 +378,7 @@ TEST_F(trace, create_domain) tracefs_count_matches(buf, REGEX_CHECK_RULE_FS(TRACE_TASK)); ASSERT_LE(1, check_count) { - TH_LOG("Expected check_rule_fs events\n%s", buf); + TH_LOG("Expected check_rule_inode events\n%s", buf); } EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_CHECK_RULE_FS(TRACE_TASK), @@ -508,9 +513,11 @@ TEST_F(trace, add_rule_invalid_fd) buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - EXPECT_EQ(0, tracefs_count_matches(buf, REGEX_ADD_RULE_FS(TRACE_TASK))) + EXPECT_EQ(0, tracefs_count_matches( + buf, REGEX_ADD_RULE_PATH_BENEATH(TRACE_TASK))) { - TH_LOG("No add_rule_fs event expected on invalid fd\n%s", buf); + TH_LOG("No add_rule_path_beneath event expected on invalid fd\n%s", + buf); } free(buf); @@ -902,10 +909,10 @@ TEST_F(trace, non_audit_visible_denial_counting) } /* - * Verifies that landlock_add_rule_net emits a trace event with the correct port - * and allowed access mask fields. + * Verifies that landlock_add_rule_net_port emits a trace event with the correct + * port and allowed access mask fields. */ -TEST_F(trace, add_rule_net_fields) +TEST_F(trace, add_rule_net_port_fields) { struct landlock_ruleset_attr ruleset_attr = { .handled_access_net = LANDLOCK_ACCESS_NET_BIND_TCP, @@ -931,9 +938,10 @@ TEST_F(trace, add_rule_net_fields) buf = tracefs_read_buf(); ASSERT_NE(NULL, buf); - EXPECT_EQ(1, tracefs_count_matches(buf, REGEX_ADD_RULE_NET(TRACE_TASK))) + EXPECT_EQ(1, tracefs_count_matches(buf, + REGEX_ADD_RULE_NET_PORT(TRACE_TASK))) { - TH_LOG("Expected 1 add_rule_net event\n%s", buf); + TH_LOG("Expected 1 add_rule_net_port event\n%s", buf); } /* @@ -941,7 +949,8 @@ TEST_F(trace, add_rule_net_fields) * (landlock_net_port_attr.port). On little-endian, htons(8080) is * 36895, so this comparison catches byte-order bugs. */ - EXPECT_EQ(0, tracefs_extract_field(buf, REGEX_ADD_RULE_NET(TRACE_TASK), + EXPECT_EQ(0, tracefs_extract_field(buf, + REGEX_ADD_RULE_NET_PORT(TRACE_TASK), "port", field, sizeof(field))); EXPECT_STREQ("8080", field); /* @@ -950,9 +959,9 @@ TEST_F(trace, add_rule_net_fields) * net access bits are unhandled because the ruleset only handles * BIND_TCP). */ - EXPECT_EQ(0, - tracefs_extract_field(buf, REGEX_ADD_RULE_NET(TRACE_TASK), - "access_rights", field, sizeof(field))); + EXPECT_EQ(0, tracefs_extract_field( + buf, REGEX_ADD_RULE_NET_PORT(TRACE_TASK), + "access_rights", field, sizeof(field))); EXPECT_STREQ("bind_tcp|connect_tcp|bind_udp|connect_send_udp", field); free(buf); diff --git a/tools/testing/selftests/nci/nci_dev.c b/tools/testing/selftests/nci/nci_dev.c index 312f84ee0444..07427fa42888 100644 --- a/tools/testing/selftests/nci/nci_dev.c +++ b/tools/testing/selftests/nci/nci_dev.c @@ -8,6 +8,7 @@ #include <stdlib.h> #include <errno.h> +#include <stdint.h> #include <string.h> #include <sys/ioctl.h> #include <fcntl.h> @@ -87,6 +88,16 @@ struct msgtemplate { char buf[MAX_MSG_SIZE]; }; +static int join_thread_status(pthread_t thread) +{ + void *thread_ret = NULL; + + if (pthread_join(thread, &thread_ret)) + return -1; + + return (int)(intptr_t)thread_ret; +} + static int create_nl_socket(void) { int fd; @@ -182,7 +193,7 @@ static int get_family_id(int sd, __u32 pid, __u32 *event_group) } ans; struct nlattr *na; int resp_len; - __u16 id; + __u16 id = 0; int len; int rc; @@ -438,13 +449,13 @@ FIXTURE_SETUP(NCI) else rc = pthread_create(&thread_t, NULL, virtual_dev_open, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_UP, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); ASSERT_EQ(status, 0); self->open_state = true; } @@ -509,12 +520,12 @@ FIXTURE_TEARDOWN(NCI) rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); ASSERT_EQ(status, 0); } @@ -585,12 +596,11 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p void *nla_start_poll_data[2] = {&dev_idx, &proto}; int nla_start_poll_len[2] = {4, 4}; pthread_t thread_t; - int status; int rc; rc = pthread_create(&thread_t, NULL, virtual_poll_start, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_mt_nla(sd, fid, pid, NFC_CMD_START_POLL, 2, nla_start_poll_type, @@ -598,19 +608,17 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p if (rc != 0) return rc; - pthread_join(thread_t, (void **)&status); - return status; + return join_thread_status(thread_t); } int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid) { pthread_t thread_t; - int status; int rc; rc = pthread_create(&thread_t, NULL, virtual_poll_stop, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_with_idx(sd, fid, pid, @@ -618,8 +626,7 @@ int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid) if (rc != 0) return rc; - pthread_join(thread_t, (void **)&status); - return status; + return join_thread_status(thread_t); } TEST_F(NCI, start_poll) @@ -830,10 +837,14 @@ int disconnect_tag(int nfc_sock, int virtual_fd) status = pthread_create(&thread_t, NULL, virtual_deactivate_proc, (void *)&virtual_fd); + if (status) + return status; close(nfc_sock); - pthread_join(thread_t, (void **)&status); - return status; + if (status) + return -1; + + return join_thread_status(thread_t); } TEST_F(NCI, t4t_tag_read) @@ -874,13 +885,13 @@ TEST_F(NCI, deinit) else rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); self->open_state = 0; ASSERT_EQ(status, 0); diff --git a/tools/testing/selftests/net/config b/tools/testing/selftests/net/config index 30d5fcb09a83..737e7e6327b3 100644 --- a/tools/testing/selftests/net/config +++ b/tools/testing/selftests/net/config @@ -118,10 +118,10 @@ CONFIG_NFT_NAT=m CONFIG_NUMA=y CONFIG_OPENVSWITCH=m CONFIG_PAGE_POOL_STATS=y -CONFIG_SYSCTL=y CONFIG_PSAMPLE=m CONFIG_RPS=y CONFIG_SYN_COOKIES=y +CONFIG_SYSCTL=y CONFIG_SYSFS=y CONFIG_TAP=m CONFIG_TCP_CONG_DCTCP=y diff --git a/tools/testing/selftests/net/nl_nlctrl.py b/tools/testing/selftests/net/nl_nlctrl.py index fe1f66dc9435..237b3d273260 100755 --- a/tools/testing/selftests/net/nl_nlctrl.py +++ b/tools/testing/selftests/net/nl_nlctrl.py @@ -9,40 +9,86 @@ from lib.py import ksft_run, ksft_exit from lib.py import ksft_eq, ksft_ge, ksft_true, ksft_in, ksft_not_in from lib.py import NetdevFamily, EthtoolFamily, NlctrlFamily +# Families we can expect to always be around, and which between them +# cover ops with a do, with a dump, and with both. +FAMILIES = ('nlctrl', 'netdev') -def getfamily_do(ctrl) -> None: - """Query a single family by name and validate its ops.""" - fam = ctrl.getfamily({'family-name': 'netdev'}) - ksft_eq(fam['family-name'], 'netdev') + +def _get_ops(ctrl, name): + """Get the ops of a family, keyed by command id.""" + fam = ctrl.getfamily({'family-name': name}) + ksft_eq(fam['family-name'], name) ksft_true(fam['family-id'] > 0) # The format of ops is quite odd, [{$idx: {"id"...}}, {$idx: {"id"...}}] # Discard the indices and re-key by command id. ops_by_id = {v['id']: v for op in fam['ops'] for v in op.values()} - ksft_eq(len(ops_by_id), len(fam['ops'])) + ksft_eq(len(ops_by_id), len(fam['ops']), + comment=f"{name} lists a command twice") + return ops_by_id + + +def _get_policy_map(ctrl, req): + """ + The policy map in the Netlink replies looks like this: + + [{'family-id': 16, 'op-policy': {'do': 0, 'dump': 0, 'op-id': 3}}, + {'family-id': 16, 'op-policy': {'dump': 1, 'op-id': 4}}, ...] + + Return the mapping: + + {3:{'do','dump'}, 4:{'dump'}} + + The policy itself is discarded here, only return which command has policy. + """ + pol_map = {} + for msg in ctrl.getpolicy(req, dump=True): + if 'op-policy' not in msg: + continue + modes = dict(msg['op-policy']) + cmd = modes.pop('op-id') + ksft_not_in(cmd, pol_map, comment=f"command {cmd} reported twice") + pol_map[cmd] = set(modes.keys()) + return pol_map + + +def getfamily_do(ctrl) -> None: + """Query single families by name and validate their ops.""" + ops = {name: _get_ops(ctrl, name) for name in FAMILIES} + + for name, ops_by_id in ops.items(): + for op in ops_by_id.values(): + # All ops in nlctrl and netdev have a policy + ksft_in('cmd-cap-haspol', op['flags'], + comment=f"{name} op {op['id']} missing haspol") + ksft_true(op['flags'] & {'cmd-cap-do', 'cmd-cap-dump'}, + comment=f"{name} op {op['id']} has no handler") - # All ops should have a policy (either do or dump has one) - for op in ops_by_id.values(): - ksft_in('cmd-cap-haspol', op['flags'], - comment=f"op {op['id']} missing haspol") + # nlctrl getfamily (id 3) does both, getpolicy (id 10) is dump-only + ksft_in('cmd-cap-do', ops['nlctrl'][3]['flags']) + ksft_in('cmd-cap-dump', ops['nlctrl'][3]['flags']) + ksft_not_in('cmd-cap-do', ops['nlctrl'][10]['flags']) + ksft_in('cmd-cap-dump', ops['nlctrl'][10]['flags']) + + netdev = ops['netdev'] # dev-get (id 1) should support both do and dump - ksft_in('cmd-cap-do', ops_by_id[1]['flags']) - ksft_in('cmd-cap-dump', ops_by_id[1]['flags']) + ksft_in('cmd-cap-do', netdev[1]['flags']) + ksft_in('cmd-cap-dump', netdev[1]['flags']) # qstats-get (id 12) is dump-only - ksft_not_in('cmd-cap-do', ops_by_id[12]['flags']) - ksft_in('cmd-cap-dump', ops_by_id[12]['flags']) + ksft_not_in('cmd-cap-do', netdev[12]['flags']) + ksft_in('cmd-cap-dump', netdev[12]['flags']) # napi-set (id 14) is do-only and requires admin - ksft_in('cmd-cap-do', ops_by_id[14]['flags']) - ksft_not_in('cmd-cap-dump', ops_by_id[14]['flags']) - ksft_in('admin-perm', ops_by_id[14]['flags']) + ksft_in('cmd-cap-do', netdev[14]['flags']) + ksft_not_in('cmd-cap-dump', netdev[14]['flags']) + ksft_in('admin-perm', netdev[14]['flags']) # Notification-only commands (dev-add/del/change-ntf etc.) must # not appear in the ops list since they have no do/dump handlers. for ntf_id in [2, 3, 4, 6, 7, 8]: - ksft_not_in(ntf_id, ops_by_id, + ksft_not_in(ntf_id, netdev, comment=f"ntf-only cmd {ntf_id} should not be in ops") @@ -103,6 +149,41 @@ def getpolicy_dump(_ctrl) -> None: comment="linkinfo-set should not have a dump policy") +def getpolicy_op_map(ctrl) -> None: + """Check the op-to-policy map consistency. Each op with 'haspol' flag + has to have a policy. The policy back-references must name only + real ops that exist, have given modes (do vs dump) and have 'haspol'. + """ + for name in FAMILIES: + ops_by_id = _get_ops(ctrl, name) + haspol = {cmd for cmd, op in ops_by_id.items() + if 'cmd-cap-haspol' in op['flags']} + + pol_map = _get_policy_map(ctrl, {'family-name': name}) + ksft_eq(set(pol_map), haspol, + comment=f"{name} policy map does not match the op list") + + # Walk the op list rather than the map, the map may be missing + # the very op we are after. Asking for a command the family does + # not have is an error, so it must not come from the map either. + for cmd in sorted(haspol): + modes = pol_map.get(cmd, set()) + + # The kernel only reports a mode the op actually has. + if 'do' in modes: + ksft_in('cmd-cap-do', ops_by_id[cmd]['flags'], + comment=f"{name} cmd {cmd} has no do") + if 'dump' in modes: + ksft_in('cmd-cap-dump', ops_by_id[cmd]['flags'], + comment=f"{name} cmd {cmd} has no dump") + + # Asking for one op builds the map in a different place in + # the kernel, it has to report what the full dump did. + single = _get_policy_map(ctrl, {'family-name': name, 'op': cmd}) + ksft_eq(single, {cmd: modes}, + comment=f"{name} cmd {cmd} policy differs from the dump") + + def getpolicy_by_op(_ctrl) -> None: """Query policy for specific ops, check attr names are resolved.""" ndev = NetdevFamily() @@ -122,6 +203,7 @@ def main() -> None: ksft_run([getfamily_do, getfamily_dump, getpolicy_dump, + getpolicy_op_map, getpolicy_by_op], args=(ctrl, )) ksft_exit() diff --git a/tools/testing/selftests/net/ovpn/common.sh b/tools/testing/selftests/net/ovpn/common.sh index 2d844eb3aa6e..5e9c81e885e6 100644 --- a/tools/testing/selftests/net/ovpn/common.sh +++ b/tools/testing/selftests/net/ovpn/common.sh @@ -136,6 +136,19 @@ ovpn_create_ns() { ip netns add "ovpn_peer${1}" } +ovpn_peer_vpn_addr() { + local peer="$1" + local file + + if [ "${OVPN_PROTO}" == "UDP" ]; then + file="${OVPN_UDP_PEERS_FILE}" + else + file="${OVPN_TCP_PEERS_FILE}" + fi + + awk -v peer="${peer}" '$1 == peer {print $NF; exit}' "${file}" +} + ovpn_setup_ns() { local peer="ovpn_peer${1}" local server_ns="ovpn_peer0" diff --git a/tools/testing/selftests/net/ovpn/ovpn-cli.c b/tools/testing/selftests/net/ovpn/ovpn-cli.c index f4effa7580c0..3b612a8a18fe 100644 --- a/tools/testing/selftests/net/ovpn/ovpn-cli.c +++ b/tools/testing/selftests/net/ovpn/ovpn-cli.c @@ -650,6 +650,26 @@ err: return ret; } +static int ovpn_nl_put_vpn_addr(struct nl_msg *msg, + const struct ovpn_ctx *ovpn) +{ + if (!ovpn->peer_ip_set) + return 0; + + switch (ovpn->peer_ip.in4.sin_family) { + case AF_INET: + return nla_put_u32(msg, OVPN_A_PEER_VPN_IPV4, + ovpn->peer_ip.in4.sin_addr.s_addr); + case AF_INET6: + return nla_put(msg, OVPN_A_PEER_VPN_IPV6, + sizeof(struct in6_addr), + &ovpn->peer_ip.in6.sin6_addr); + default: + fprintf(stderr, "Invalid family for peer address\n"); + return -EAFNOSUPPORT; + } +} + static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp) { struct nlattr *attr; @@ -691,22 +711,9 @@ static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp) } } - if (ovpn->peer_ip_set) { - switch (ovpn->peer_ip.in4.sin_family) { - case AF_INET: - NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_VPN_IPV4, - ovpn->peer_ip.in4.sin_addr.s_addr); - break; - case AF_INET6: - NLA_PUT(ctx->nl_msg, OVPN_A_PEER_VPN_IPV6, - sizeof(struct in6_addr), - &ovpn->peer_ip.in6.sin6_addr); - break; - default: - fprintf(stderr, "Invalid family for peer address\n"); - goto nla_put_failure; - } - } + ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn); + if (ret) + goto nla_put_failure; nla_nest_end(ctx->nl_msg, attr); @@ -732,6 +739,10 @@ static int ovpn_set_peer(struct ovpn_ctx *ovpn) ovpn->keepalive_interval); NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_KEEPALIVE_TIMEOUT, ovpn->keepalive_timeout); + + ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn); + if (ret) + goto nla_put_failure; nla_nest_end(ctx->nl_msg, attr); ret = ovpn_nl_msg_send(ctx, NULL); @@ -1730,13 +1741,14 @@ static void usage(const char *cmd) fprintf(stderr, "\tmark: socket FW mark value\n"); fprintf(stderr, - "* set_peer <iface> <peer_id> <keepalive_interval> <keepalive_timeout>: set peer attributes\n"); + "* set_peer <iface> <peer_id> <keepalive_interval> <keepalive_timeout> [vpnaddr]: set peer attributes\n"); fprintf(stderr, "\tiface: ovpn interface name\n"); fprintf(stderr, "\tpeer_id: peer ID of the peer to modify\n"); fprintf(stderr, "\tkeepalive_interval: interval for sending ping messages\n"); fprintf(stderr, "\tkeepalive_timeout: time after which a peer is timed out\n"); + fprintf(stderr, "\tvpnaddr: peer VPN IP\n"); fprintf(stderr, "* del_peer <iface> <peer_id>: delete peer\n"); fprintf(stderr, "\tiface: ovpn interface name\n"); @@ -2090,6 +2102,8 @@ static int ovpn_run_cmd(struct ovpn_ctx *ovpn) return ret; ret = ovpn_new_peer(ovpn, false); + if (ret < 0) + return ret; ovpn_waitbg(); break; case CMD_NEW_MULTI_PEER: @@ -2331,6 +2345,12 @@ static int ovpn_parse_cmd_args(struct ovpn_ctx *ovpn, int argc, char *argv[]) "keepalive interval value out of range\n"); return -1; } + + if (argc > 6) { + ret = ovpn_parse_remote(ovpn, NULL, NULL, argv[6]); + if (ret < 0) + return -1; + } break; case CMD_DEL_PEER: if (argc < 4) diff --git a/tools/testing/selftests/net/ovpn/test.sh b/tools/testing/selftests/net/ovpn/test.sh index 9b5610837032..392109d5e14e 100755 --- a/tools/testing/selftests/net/ovpn/test.sh +++ b/tools/testing/selftests/net/ovpn/test.sh @@ -56,6 +56,76 @@ ovpn_prepare_network() { done } +ovpn_new_test_peer() { + local peer_id="$1" + + shift + ip netns exec ovpn_peer0 "${OVPN_CLI}" new_peer tun0 \ + "${peer_id}" none 65000 10.10.1.2 1 "$@" +} + +ovpn_set_peer_vpn_addr() { + ip netns exec ovpn_peer0 "${OVPN_CLI}" set_peer tun0 \ + "$1" 60 120 "$2" +} + +ovpn_run_vpn_addr_validation() { + local addr + local peer1_addr4 + local test_peer_id=$((OVPN_NUM_PEERS + 1)) + local test_peer_addr6="2001:db8::2" + # Do not include 0.0.0.0 or :: here. They are invalid on creation, but + # clear one address family on update and are valid if the other remains. + local -a invalid_addrs=( + "127.0.0.1" + "224.0.0.1" + "255.255.255.255" + "::1" + "::192.0.2.1" + "::ffff:192.0.2.1" + "ff02::1" + ) + + peer1_addr4=$(ovpn_peer_vpn_addr 1) + + ovpn_cmd_fail "reject peer without VPN address" \ + ovpn_new_test_peer "${test_peer_id}" + + for addr in "0.0.0.0" "::" "${invalid_addrs[@]}"; do + ovpn_cmd_fail "reject new peer VPN address ${addr}" \ + ovpn_new_test_peer "${test_peer_id}" "${addr}" + done + + ovpn_cmd_fail "reject duplicate IPv4 address on peer creation" \ + ovpn_new_test_peer "${test_peer_id}" "${peer1_addr4}" + ovpn_cmd_fail "reject clearing the last peer VPN address" \ + ovpn_set_peer_vpn_addr 1 0.0.0.0 + + for addr in "${invalid_addrs[@]}"; do + ovpn_cmd_fail "reject updated peer VPN address ${addr}" \ + ovpn_set_peer_vpn_addr 1 "${addr}" + done + + ovpn_cmd_fail "reject duplicate IPv4 address on peer update" \ + ovpn_set_peer_vpn_addr 2 "${peer1_addr4}" + + ovpn_cmd_ok "add peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 "${test_peer_addr6}" + ovpn_cmd_fail "reject duplicate IPv6 address on peer creation" \ + ovpn_new_test_peer "${test_peer_id}" "${test_peer_addr6}" + ovpn_cmd_fail "reject duplicate IPv6 address on peer update" \ + ovpn_set_peer_vpn_addr 2 "${test_peer_addr6}" + + ovpn_cmd_ok "clear peer IPv4 address" \ + ovpn_set_peer_vpn_addr 1 0.0.0.0 + ovpn_cmd_fail "reject clearing the remaining peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 :: + ovpn_cmd_ok "restore peer IPv4 address" \ + ovpn_set_peer_vpn_addr 1 "${peer1_addr4}" + ovpn_cmd_ok "clear peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 :: +} + ovpn_run_basic_traffic() { local p local header1 @@ -293,15 +363,16 @@ trap ovpn_stage_err ERR ktap_print_header if [ "${OVPN_FLOAT}" == "1" ]; then - ktap_set_plan 13 + ktap_set_plan 14 else - ktap_set_plan 12 + ktap_set_plan 13 fi ovpn_cleanup modprobe -q ovpn || true ovpn_run_stage "setup network topology" ovpn_prepare_network +ovpn_run_stage "validate peer VPN addresses" ovpn_run_vpn_addr_validation ovpn_run_stage "run baseline data traffic" ovpn_run_basic_traffic ovpn_run_stage "run LAN traffic behind peer1" ovpn_run_lan_traffic [ "${OVPN_FLOAT}" == "1" ] && ovpn_run_stage "run floating peer checks" \ diff --git a/tools/testing/selftests/net/packetdrill/config b/tools/testing/selftests/net/packetdrill/config index 83dde525c53c..b7df8bf30920 100644 --- a/tools/testing/selftests/net/packetdrill/config +++ b/tools/testing/selftests/net/packetdrill/config @@ -4,8 +4,8 @@ CONFIG_IPV6=y CONFIG_NET_NS=y CONFIG_NET_SCH_FIFO=y CONFIG_NET_SCH_FQ=y -CONFIG_SYSCTL=y CONFIG_SYN_COOKIES=y +CONFIG_SYSCTL=y CONFIG_TCP_CONG_CUBIC=y CONFIG_TCP_MD5SIG=y CONFIG_TUN=y diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile index 3cfe90e0f34f..4e06d0baaeec 100644 --- a/tools/testing/selftests/sched_ext/Makefile +++ b/tools/testing/selftests/sched_ext/Makefile @@ -164,10 +164,12 @@ all_test_bpfprogs := $(foreach prog,$(wildcard *.bpf.c),$(INCLUDE_DIR)/$(patsubs auto-test-targets := \ create_dsq \ dequeue \ + dequeue_iter \ enq_last_no_enq_fails \ ddsp_bogus_dsq_fail \ ddsp_vtimelocal_fail \ dsp_local_on \ + enable_cmask \ enq_select_cpu \ exit \ hotplug \ diff --git a/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c b/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c new file mode 100644 index 000000000000..76c2c71a90b6 --- /dev/null +++ b/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * ops.dequeue() of this scheduler iterates the user DSQ it consumes + * tasks from with bpf_iter_scx_dsq, which takes the DSQ lock. + * On a kernel that still runs ops.dequeue() with that lock held, the + * iteration self-deadlocks the CPU - this test wedges the system on + * unfixed kernels instead of failing cleanly. + * + * Copyright (c) 2026 fangqiurong <fangqiurong@kylinos.cn> + */ + +#include <scx/common.bpf.h> + +char _license[] SEC("license") = "GPL"; + +UEI_DEFINE(uei); + +#define TEST_DSQ_ID 1000 + +u64 dq_count; + +s32 BPF_STRUCT_OPS_SLEEPABLE(dequeue_iter_init) +{ + return scx_bpf_create_dsq(TEST_DSQ_ID, -1); +} + +s32 BPF_STRUCT_OPS(dequeue_iter_select_cpu, struct task_struct *p, + s32 prev_cpu, u64 wake_flags) +{ + return prev_cpu; +} + +void BPF_STRUCT_OPS(dequeue_iter_enqueue, struct task_struct *p, u64 enq_flags) +{ + scx_bpf_dsq_insert(p, TEST_DSQ_ID, SCX_SLICE_DFL, enq_flags); +} + +void BPF_STRUCT_OPS(dequeue_iter_dispatch, s32 cpu, struct task_struct *task) +{ + scx_bpf_dsq_move_to_local(TEST_DSQ_ID, 0); +} + +void BPF_STRUCT_OPS(dequeue_iter_dequeue, struct task_struct *p, u64 deq_flags) +{ + struct bpf_iter_scx_dsq it; + struct task_struct *t; + + if (!bpf_iter_scx_dsq_new(&it, TEST_DSQ_ID, 0)) { + while ((t = bpf_iter_scx_dsq_next(&it))) + ; + } + bpf_iter_scx_dsq_destroy(&it); + + __sync_fetch_and_add(&dq_count, 1); +} + +void BPF_STRUCT_OPS(dequeue_iter_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); + scx_bpf_destroy_dsq(TEST_DSQ_ID); +} + +SEC(".struct_ops.link") +struct sched_ext_ops dequeue_iter_ops = { + .init = (void *)dequeue_iter_init, + .select_cpu = (void *)dequeue_iter_select_cpu, + .enqueue = (void *)dequeue_iter_enqueue, + .dispatch = (void *)dequeue_iter_dispatch, + .dequeue = (void *)dequeue_iter_dequeue, + .exit = (void *)dequeue_iter_exit, + .timeout_ms = 1000U, + .name = "dequeue_iter", +}; diff --git a/tools/testing/selftests/sched_ext/dequeue_iter.c b/tools/testing/selftests/sched_ext/dequeue_iter.c new file mode 100644 index 000000000000..f60711bf3b39 --- /dev/null +++ b/tools/testing/selftests/sched_ext/dequeue_iter.c @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (c) 2026 fangqiurong <fangqiurong@kylinos.cn> + */ +#include <bpf/bpf.h> +#include <scx/common.h> +#include <time.h> +#include <unistd.h> +#include "dequeue_iter.bpf.skel.h" +#include "scx_test.h" + +#define DQ_TARGET 10 +#define DQ_DEADLINE_MS 3000 + +static unsigned long long now_ms(void) +{ + struct timespec ts; + + clock_gettime(CLOCK_MONOTONIC, &ts); + + return ts.tv_sec * 1000ULL + ts.tv_nsec / 1000000; +} + +static enum scx_test_status setup(void **ctx) +{ + struct dequeue_iter *skel; + + skel = dequeue_iter__open(); + SCX_FAIL_IF(!skel, "Failed to open"); + SCX_ENUM_INIT(skel); + SCX_FAIL_IF(dequeue_iter__load(skel), "Failed to load skel"); + + *ctx = skel; + + return SCX_TEST_PASS; +} + +static enum scx_test_status run(void *ctx) +{ + struct dequeue_iter *skel = ctx; + struct bpf_link *link; + unsigned long long end; + + link = bpf_map__attach_struct_ops(skel->maps.dequeue_iter_ops); + SCX_FAIL_IF(!link, "Failed to attach scheduler"); + + end = now_ms() + DQ_DEADLINE_MS; + while (skel->bss->dq_count < DQ_TARGET && !UEI_EXITED(skel, uei) && + now_ms() < end) + usleep(100); + + bpf_link__destroy(link); + + SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG)); + + if (skel->bss->dq_count < DQ_TARGET) { + SCX_ERR("ops.dequeue() fired only %llu times", + (unsigned long long)skel->bss->dq_count); + return SCX_TEST_FAIL; + } + + return SCX_TEST_PASS; +} + +static void cleanup(void *ctx) +{ + struct dequeue_iter *skel = ctx; + + dequeue_iter__destroy(skel); +} + +struct scx_test dequeue_iter = { + .name = "dequeue_iter", + .description = "Verify ops.dequeue() can iterate its source user DSQ", + .setup = setup, + .run = run, + .cleanup = cleanup, +}; +REGISTER_SCX_TEST(&dequeue_iter) diff --git a/tools/testing/selftests/sched_ext/enable_cmask.bpf.c b/tools/testing/selftests/sched_ext/enable_cmask.bpf.c new file mode 100644 index 000000000000..0068f3c7ab3c --- /dev/null +++ b/tools/testing/selftests/sched_ext/enable_cmask.bpf.c @@ -0,0 +1,217 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A cid-form scheduler checking the cmask cid-form ops.enable() receives: the + * header, every cid bit against p->cpus_ptr, and that set_cmask() follows with + * the same mask before set_weight() and before the task first becomes runnable, + * and never runs before enable(). + * + * Copyright (c) 2026 Tejun Heo <tj@kernel.org> + */ +#include <scx/common.bpf.h> + +char _license[] SEC("license") = "GPL"; + +struct { + __uint(type, BPF_MAP_TYPE_ARENA); + __uint(map_flags, BPF_F_MMAPABLE); + __uint(max_entries, 1 << 16); +} arena SEC(".maps"); + +struct task_ctx { + u64 enable_fp; /* fingerprint of the mask enable() received */ + bool enabled; + bool pending; /* enable() ran, the initial set_cmask() hasn't */ +}; + +struct { + __uint(type, BPF_MAP_TYPE_TASK_STORAGE); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, int); + __type(value, struct task_ctx); +} task_ctx_stor SEC(".maps"); + +/* details of a cid bit mismatch, filled by check_mask() */ +struct mask_mismatch { + s32 cid; + bool want; + bool got; +}; + +u64 nr_enable, nr_initial_set_cmask, nr_set_cmask, nr_set_weight; + +UEI_DEFINE(uei); + +static struct task_ctx *lookup_task_ctx(struct task_struct *p) +{ + struct task_ctx *tctx; + + tctx = bpf_task_storage_get(&task_ctx_stor, p, 0, 0); + if (!tctx) + scx_bpf_error("task_ctx lookup failed for %s[%d]", p->comm, p->pid); + return tctx; +} + +/* + * Verify @m's header and every cid bit against @p's cpumask and fingerprint the + * bits into @fp. Return 0 on success, -EINVAL on a bad header, -ENOENT on a cid + * without a cpu and -EIO on a bit mismatch with the details in @mm. + */ +static int check_mask(struct task_struct *p, const struct scx_cmask __arena *m, u64 *fp, + struct mask_mismatch *mm) +{ + u32 nr_cids = scx_bpf_nr_cids(); + u64 h = 0; + s32 cid; + + if (m->base || m->nr_cids != nr_cids || m->alloc_words != CMASK_NR_WORDS(nr_cids)) + return -EINVAL; + + bpf_for(cid, 0, nr_cids) { + bool want, got; + s32 cpu; + + cpu = scx_bpf_cid_to_cpu(cid); + if (cpu < 0) + return -ENOENT; + want = bpf_cpumask_test_cpu(cpu, p->cpus_ptr); + got = cmask_test(cid, m); + if (want != got) { + mm->cid = cid; + mm->want = want; + mm->got = got; + return -EIO; + } + h = h * 31 + got; + } + + *fp = h; + return 0; +} + +s32 BPF_STRUCT_OPS_SLEEPABLE(enable_cmask_init_task, struct task_struct *p, + struct scx_init_task_args *args) +{ + if (!bpf_task_storage_get(&task_ctx_stor, p, 0, BPF_LOCAL_STORAGE_GET_F_CREATE)) + return -ENOMEM; + return 0; +} + +void BPF_STRUCT_OPS(enable_cmask_enable, struct task_struct *p, struct scx_enable_args *args) +{ + struct scx_cmask __arena *m = (struct scx_cmask __arena *)args->cmask_arena_addr; + struct mask_mismatch mm = {}; + struct task_ctx *tctx; + int ret; + + asm volatile("" :: "r"(&arena)); + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + __sync_fetch_and_add(&nr_enable, 1); + if (tctx->enabled || tctx->pending) { + scx_bpf_error("enable: %s[%d] enabled twice", p->comm, p->pid); + return; + } + + ret = check_mask(p, m, &tctx->enable_fp, &mm); + if (ret) { + scx_bpf_error("enable: %s[%d] cmask check failed %d cid=%d want=%d got=%d", + p->comm, p->pid, ret, mm.cid, mm.want, mm.got); + return; + } + tctx->enabled = true; + tctx->pending = true; +} + +void BPF_STRUCT_OPS(enable_cmask_set_cmask, struct task_struct *p, + struct scx_cmask __arena *m) +{ + struct mask_mismatch mm = {}; + struct task_ctx *tctx; + u64 fp; + int ret; + + asm volatile("" :: "r"(&arena)); + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + __sync_fetch_and_add(&nr_set_cmask, 1); + if (!tctx->enabled) { + scx_bpf_error("set_cmask: %s[%d] not enabled", p->comm, p->pid); + return; + } + + ret = check_mask(p, m, &fp, &mm); + if (ret) { + scx_bpf_error("set_cmask: %s[%d] cmask check failed %d cid=%d want=%d got=%d", + p->comm, p->pid, ret, mm.cid, mm.want, mm.got); + return; + } + + if (tctx->pending) { + if (fp != tctx->enable_fp) { + scx_bpf_error("set_cmask: %s[%d] initial mask differs from enable()", + p->comm, p->pid); + return; + } + tctx->pending = false; + __sync_fetch_and_add(&nr_initial_set_cmask, 1); + } +} + +void BPF_STRUCT_OPS(enable_cmask_set_weight, struct task_struct *p, u32 weight) +{ + struct task_ctx *tctx; + + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + __sync_fetch_and_add(&nr_set_weight, 1); + if (tctx->pending) + scx_bpf_error("set_weight: %s[%d] before the initial set_cmask()", p->comm, + p->pid); +} + +void BPF_STRUCT_OPS(enable_cmask_runnable, struct task_struct *p, u64 enq_flags) +{ + struct task_ctx *tctx; + + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + if (tctx->pending) + scx_bpf_error("runnable: %s[%d] before the initial set_cmask()", p->comm, + p->pid); +} + +void BPF_STRUCT_OPS(enable_cmask_disable, struct task_struct *p) +{ + struct task_ctx *tctx; + + tctx = lookup_task_ctx(p); + if (!tctx) + return; + + tctx->enabled = false; + tctx->pending = false; +} + +void BPF_STRUCT_OPS(enable_cmask_exit, struct scx_exit_info *ei) +{ + UEI_RECORD(uei, ei); +} + +SCX_OPS_CID_DEFINE(enable_cmask_ops, + .init_task = (void *)enable_cmask_init_task, + .enable = (void *)enable_cmask_enable, + .set_cmask = (void *)enable_cmask_set_cmask, + .set_weight = (void *)enable_cmask_set_weight, + .runnable = (void *)enable_cmask_runnable, + .disable = (void *)enable_cmask_disable, + .exit = (void *)enable_cmask_exit, + .flags = SCX_OPS_SWITCH_PARTIAL, + .name = "enable_cmask"); diff --git a/tools/testing/selftests/sched_ext/enable_cmask.c b/tools/testing/selftests/sched_ext/enable_cmask.c new file mode 100644 index 000000000000..556bcad4431d --- /dev/null +++ b/tools/testing/selftests/sched_ext/enable_cmask.c @@ -0,0 +1,138 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 Tejun Heo <tj@kernel.org> */ +#define _GNU_SOURCE +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <time.h> +#include <unistd.h> +#include <sys/wait.h> +#include <bpf/bpf.h> +#include <scx/common.h> +#include "enable_cmask.bpf.skel.h" +#include "scx_test.h" + +#define SCHED_EXT 7 +#define NR_CHILDREN 8 +#define MAX_CPUS 1024 + +static int cpus[MAX_CPUS]; +static int nr_cpus; + +static void spin_ms(int ms) +{ + struct timespec start, now; + + clock_gettime(CLOCK_MONOTONIC, &start); + do { + clock_gettime(CLOCK_MONOTONIC, &now); + } while ((now.tv_sec - start.tv_sec) * 1000 + + (now.tv_nsec - start.tv_nsec) / 1000000 < ms); +} + +static int pin(pid_t pid, int idx) +{ + cpu_set_t set; + + CPU_ZERO(&set); + CPU_SET(cpus[idx % nr_cpus], &set); + return sched_setaffinity(pid, sizeof(set), &set); +} + +/* + * Pin, switch to SCHED_EXT for a class-switch enable, fork a grandchild that + * inherits the policy for a fork-path enable, then change affinity a few times + * while running for set_cmask() on live tasks. + */ +static int child(int idx) +{ + struct sched_param param = {}; + int i, status; + pid_t pid; + + if (pin(0, idx) || sched_setscheduler(0, SCHED_EXT, ¶m)) + return 1; + + pid = fork(); + if (pid < 0) + return 1; + if (!pid) { + spin_ms(20); + return 0; + } + + for (i = 1; i <= 4; i++) { + if (pin(0, idx + i)) + return 1; + spin_ms(5); + } + + return waitpid(pid, &status, 0) == pid && !status ? 0 : 1; +} + +static enum scx_test_status run(void *ctx) +{ + struct enable_cmask *skel; + struct bpf_link *link; + pid_t pids[NR_CHILDREN]; + cpu_set_t set; + int i, status, failed = 0; + + if (!__COMPAT_struct_has_field("scx_enable_args", "cmask_arena_addr")) + return SCX_TEST_SKIP; + + SCX_FAIL_IF(sched_getaffinity(0, sizeof(set), &set), "Failed to read affinity"); + for (i = 0; i < MAX_CPUS && i < CPU_SETSIZE; i++) + if (CPU_ISSET(i, &set)) + cpus[nr_cpus++] = i; + if (nr_cpus < 2) + return SCX_TEST_SKIP; + + skel = enable_cmask__open(); + SCX_FAIL_IF(!skel, "Failed to open"); + SCX_ENUM_INIT(skel); + SCX_FAIL_IF(enable_cmask__load(skel), "Failed to load skel"); + + link = bpf_map__attach_struct_ops(skel->maps.enable_cmask_ops); + SCX_FAIL_IF(!link, "Failed to attach struct_ops"); + + for (i = 0; i < NR_CHILDREN; i++) { + pids[i] = fork(); + SCX_FAIL_IF(pids[i] < 0, "Failed to fork"); + if (!pids[i]) + exit(child(i)); + } + + /* affinity changes from the outside race with the children's own */ + for (i = 0; i < NR_CHILDREN; i++) + pin(pids[i], i + NR_CHILDREN); + + for (i = 0; i < NR_CHILDREN; i++) { + if (waitpid(pids[i], &status, 0) != pids[i] || status) + failed++; + } + + bpf_link__destroy(link); + + SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG)); + SCX_EQ(failed, 0); + SCX_GE(skel->bss->nr_enable, 2 * NR_CHILDREN); + SCX_EQ(skel->bss->nr_initial_set_cmask, skel->bss->nr_enable); + SCX_GT(skel->bss->nr_set_cmask, skel->bss->nr_initial_set_cmask); + SCX_GE(skel->bss->nr_set_weight, skel->bss->nr_enable); + printf("enable=%lu initial_set_cmask=%lu set_cmask=%lu set_weight=%lu\n", + (unsigned long)skel->bss->nr_enable, + (unsigned long)skel->bss->nr_initial_set_cmask, + (unsigned long)skel->bss->nr_set_cmask, + (unsigned long)skel->bss->nr_set_weight); + + enable_cmask__destroy(skel); + return SCX_TEST_PASS; +} + +struct scx_test enable_cmask = { + .name = "enable_cmask", + .description = "Check the cid-form ops.enable() cmask and the set_cmask() after it", + .run = run, +}; +REGISTER_SCX_TEST(&enable_cmask) diff --git a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json index e2b03f2b5e89..edc5148a8d97 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json +++ b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json @@ -376,5 +376,53 @@ "teardown": [ "$TC qdisc del dev $DUMMY clsact" ] + }, + { + "id": "35fc", + "name": "u32 manual table then auto table: auto allocation must not alias a live manual handle", + "category": [ + "filter", + "u32" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DEV1 ingress", + "$TC filter add dev $DEV1 ingress protocol ip pref 1 handle 801: u32 divisor 16" + ], + "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 2 u32 divisor 16", + "expExitCode": "0", + "verifyCmd": "$TC -d filter show dev $DEV1 ingress", + "matchPattern": "fh 801:", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress" + ] + }, + { + "id": "a6e8", + "name": "u32 manual table add/del does not leak its idr entry (re-adding the same handle succeeds)", + "category": [ + "filter", + "u32" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DEV1 ingress", + "$TC filter add dev $DEV1 ingress protocol ip pref 1 u32 divisor 16", + "$TC filter add dev $DEV1 ingress protocol ip pref 5 handle 901: u32 divisor 1", + "$TC filter del dev $DEV1 ingress protocol ip pref 5 handle 901: u32" + ], + "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 6 handle 901: u32 divisor 1", + "expExitCode": "0", + "verifyCmd": "$TC -d filter show dev $DEV1 ingress", + "matchPattern": "fh 901: ht divisor 1", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress" + ] } ] diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json index c98c339424d4..4f6bbb8b57f9 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json +++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json @@ -169,5 +169,39 @@ "teardown": [ "$TC qdisc del dev $DUMMY handle 1: root" ] + }, + { + "id": "8c39", + "name": "HFSC classify walk still reaches leaf after lateral drift", + "category": [ + "qdisc", + "hfsc" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "ip link set lo up", + "$TC qdisc add dev lo handle 1: root hfsc default 30", + "$TC class add dev lo parent 1: classid 1:1 hfsc rt m2 100kbit", + "$TC class add dev lo parent 1:1 classid 1:10 hfsc rt m2 50kbit", + "$TC class add dev lo parent 1: classid 1:2 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1: protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:1", + "$TC filter add dev lo parent 1:1 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:2", + "$TC class add dev lo parent 1:2 classid 1:20 hfsc rt m2 10kbit", + "$TC class add dev lo parent 1: classid 1:3 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1:2 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:3", + "$TC class add dev lo parent 1:3 classid 1:30 hfsc rt m2 10kbit", + "$TC class add dev lo parent 1:3 classid 1:31 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1:3 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:31" + ], + "cmdUnderTest": "ping -n -c 10 -W 1 127.0.0.1", + "expExitCode": "0", + "verifyCmd": "$TC -s class show dev lo", + "matchPattern": "class hfsc 1:31 parent 1:3 rt[^\\n]*\\n Sent [0-9]+ bytes [1-9][0-9]* pkt", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev lo handle 1: root" + ] } ] |
