diff options
| author | Mark Brown <broonie@kernel.org> | 2026-10-01 14:33:57 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-10-01 14:33:57 +0100 |
| commit | 3cae905fd2a4aa08e855f5882df66ea84d589cd4 (patch) | |
| tree | 09df1ee63b8e9271bceb996fa6b2636e221dcaad /net | |
| parent | da68210aabf496d993ec2376b5cf2907874f8601 (diff) | |
| parent | 6a75c73eebd4d497ded7d08b47894f9ddbebb5a9 (diff) | |
| download | linux-next-3cae905fd2a4aa08e855f5882df66ea84d589cd4.tar.gz linux-next-3cae905fd2a4aa08e855f5882df66ea84d589cd4.zip | |
Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git
# Conflicts:
# arch/arm64/net/bpf_jit_comp.c
# arch/x86/net/bpf_jit_comp.c
# mm/internal.h
Diffstat (limited to 'net')
| -rw-r--r-- | net/core/filter.c | 33 | ||||
| -rw-r--r-- | net/ipv4/Makefile | 1 | ||||
| -rw-r--r-- | net/ipv4/af_inet.c | 1 | ||||
| -rw-r--r-- | net/ipv4/bpf_tcp_ca.c | 16 | ||||
| -rw-r--r-- | net/ipv4/bpf_tcp_ops.c | 326 | ||||
| -rw-r--r-- | net/ipv4/tcp.c | 1 | ||||
| -rw-r--r-- | net/ipv4/tcp_bpf.c | 1 | ||||
| -rw-r--r-- | net/ipv4/tcp_input.c | 17 | ||||
| -rw-r--r-- | net/ipv4/tcp_output.c | 98 | ||||
| -rw-r--r-- | net/ipv4/tcp_timer.c | 1 | ||||
| -rw-r--r-- | net/sched/bpf_qdisc.c | 2 |
11 files changed, 461 insertions, 36 deletions
diff --git a/net/core/filter.c b/net/core/filter.c index 70dc621672f2..ba536be2915f 100644 --- a/net/core/filter.c +++ b/net/core/filter.c @@ -8052,17 +8052,14 @@ static const u8 *bpf_search_tcp_opt(const u8 *op, const u8 *opend, return ERR_PTR(-ENOMSG); } -BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, - void *, search_res, u32, len, u64, flags) +int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + void *search_res, u32 len, u64 flags) { bool eol, load_syn = flags & BPF_LOAD_HDR_OPT_TCP_SYN; const u8 *op, *opend, *magic, *search = search_res; u8 search_kind, search_len, copy_len, magic_len; int ret; - if (!is_locked_tcp_sock_ops(bpf_sock)) - return -EOPNOTSUPP; - /* 2 byte is the minimal option len except TCPOPT_NOP and * TCPOPT_EOL which are useless for the bpf prog to learn * and this helper disallow loading them also. @@ -8123,6 +8120,15 @@ BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, return ret; } +BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, + void *, search_res, u32, len, u64, flags) +{ + if (!is_locked_tcp_sock_ops(bpf_sock)) + return -EOPNOTSUPP; + + return __bpf_sock_ops_load_hdr_opt(bpf_sock, search_res, len, flags); +} + static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = { .func = bpf_sock_ops_load_hdr_opt, .gpl_only = false, @@ -8133,17 +8139,14 @@ static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = { .arg4_type = ARG_ANYTHING, }; -BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, - const void *, from, u32, len, u64, flags) +int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + const void *from, u32 len, u64 flags) { u8 new_kind, new_kind_len, magic_len = 0, *opend; const u8 *op, *new_op, *magic = NULL; struct sk_buff *skb; bool eol; - if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB) - return -EPERM; - if (len < 2 || flags) return -EINVAL; @@ -8201,6 +8204,15 @@ BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, return 0; } +BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, + const void *, from, u32, len, u64, flags) +{ + if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB) + return -EPERM; + + return __bpf_sock_ops_store_hdr_opt(bpf_sock, from, len, flags); +} + static const struct bpf_func_proto bpf_sock_ops_store_hdr_opt_proto = { .func = bpf_sock_ops_store_hdr_opt, .gpl_only = false, @@ -12878,6 +12890,7 @@ static int __init bpf_kfunc_init(void) ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_LWT_SEG6LOCAL, &bpf_kfunc_set_skb); ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_NETFILTER, &bpf_kfunc_set_skb); ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING, &bpf_kfunc_set_skb); + ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &bpf_kfunc_set_skb); ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_SCHED_CLS, &bpf_kfunc_set_skb_meta); ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_SCHED_ACT, &bpf_kfunc_set_skb_meta); ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_XDP, &bpf_kfunc_set_xdp); diff --git a/net/ipv4/Makefile b/net/ipv4/Makefile index 06e21c26b76f..afbac63d1cb4 100644 --- a/net/ipv4/Makefile +++ b/net/ipv4/Makefile @@ -70,6 +70,7 @@ obj-$(CONFIG_TCP_AO) += tcp_ao.o ifeq ($(CONFIG_BPF_JIT),y) obj-$(CONFIG_BPF_SYSCALL) += bpf_tcp_ca.o +obj-$(CONFIG_CGROUP_BPF) += bpf_tcp_ops.o endif ifdef CONFIG_GCOV_PROFILE_NETFILTER diff --git a/net/ipv4/af_inet.c b/net/ipv4/af_inet.c index 8e5fc11c5f2c..e94154c0ac89 100644 --- a/net/ipv4/af_inet.c +++ b/net/ipv4/af_inet.c @@ -227,6 +227,7 @@ int __inet_listen_sk(struct sock *sk, int backlog) return err; tcp_call_bpf(sk, BPF_SOCK_OPS_TCP_LISTEN_CB, 0, NULL); + bpf_tcp_ops_call(listen, sk); } return 0; } diff --git a/net/ipv4/bpf_tcp_ca.c b/net/ipv4/bpf_tcp_ca.c index 9deed2244c2d..2a71dc81562f 100644 --- a/net/ipv4/bpf_tcp_ca.c +++ b/net/ipv4/bpf_tcp_ca.c @@ -340,6 +340,22 @@ static struct bpf_struct_ops bpf_tcp_congestion_ops = { .validate = bpf_tcp_ca_validate, .name = "tcp_congestion_ops", .cfi_stubs = &__bpf_ops_tcp_congestion_ops, + /* The struct_ops's function may switch to another struct_ops. + * + * For example, bpf_tcp_cc_x->init() may switch to + * another tcp_cc_y by calling + * setsockopt(TCP_CONGESTION, "tcp_cc_y"). + * During the switch, bpf_struct_ops_put(tcp_cc_x) is called + * and its refcount may reach 0 which then free its + * trampoline image while tcp_cc_x is still running. + * + * A vanilla rcu gp is to wait for all bpf-tcp-cc prog + * to finish. bpf-tcp-cc prog is non sleepable. + * A rcu_tasks gp is to wait for the last few insn + * in the tramopline image to finish before releasing + * the trampoline image. + */ + .free_after_tasks_rcu_gp = true, .owner = THIS_MODULE, }; diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c new file mode 100644 index 000000000000..681fed642999 --- /dev/null +++ b/net/ipv4/bpf_tcp_ops.c @@ -0,0 +1,326 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */ + +#include <linux/bpf.h> +#include <linux/btf_ids.h> +#include <linux/bpf_verifier.h> +#include <linux/filter.h> +#include <net/bpf_sk_storage.h> +#include <net/tcp.h> + +static int timeout_init_stub(struct sock *sk, struct request_sock *req__nullable) +{ + struct bpf_tramp_run_ctx *ctx = + container_of(current->bpf_ctx, struct bpf_tramp_run_ctx, run_ctx); + + return ctx->retval; +} + +static int rwnd_init_stub(struct sock *sk, struct request_sock *req__nullable) +{ + struct bpf_tramp_run_ctx *ctx = + container_of(current->bpf_ctx, struct bpf_tramp_run_ctx, run_ctx); + + return ctx->retval; +} + +static void active_established_stub(struct sock *sk, struct sk_buff *skb__nullable) +{ +} + +static void passive_established_stub(struct sock *sk, struct sk_buff *skb) +{ +} + +static void rto_stub(struct sock *sk) +{ +} + +static void rtt_stub(struct sock *sk, long mrtt, u32 srtt) +{ +} + +static void set_state_stub(struct sock *sk, int state) +{ +} + +static void retrans_stub(struct sock *sk, struct sk_buff *skb, int err) +{ +} + +static void connect_stub(struct sock *sk) +{ +} + +static void listen_stub(struct sock *sk) +{ +} + +static void parse_hdr_stub(struct sock *sk, struct sk_buff *skb) +{ +} + +static void hdr_opt_len_stub(struct sock *sk, struct sk_buff *skb__nullable, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + unsigned int *remaining) +{ +} + +static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + u32 opt_off) +{ +} + +static struct bpf_tcp_ops __bpf_tcp_ops = { + .timeout_init = timeout_init_stub, + .rwnd_init = rwnd_init_stub, + .active_established = active_established_stub, + .passive_established = passive_established_stub, + .rto = rto_stub, + .rtt = rtt_stub, + .set_state = set_state_stub, + .retrans = retrans_stub, + .connect = connect_stub, + .listen = listen_stub, + .parse_hdr = parse_hdr_stub, + .hdr_opt_len = hdr_opt_len_stub, + .write_hdr_opt = write_hdr_opt_stub, +}; + +BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + u32 opt_off = args[5]; + u8 *op, *opend; + + /* + * bpf_tcp_ops does not keep track of the end of the written TCP header + * options, so search for it every time the helper is called. The free + * space is NOP-filled, so a TCPOPT_NOP ends the search rather than being + * skipped as in a normal option walk in sockops. + */ + op = skb->data + opt_off; + opend = skb->data + tcp_hdrlen(skb); + while (op < opend && *op != TCPOPT_NOP) { + if (*op == TCPOPT_EOL || op + 1 >= opend || op[1] < 2) + break; + op += op[1]; + } + + sock_ops.skb = skb; + sock_ops.skb_data_end = op; + sock_ops.remaining_opt_len = opend - op; + + return __bpf_sock_ops_store_hdr_opt(&sock_ops, from, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_store_hdr_opt_proto = { + .func = bpf_tcp_ops_store_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_RDONLY, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_4(bpf_tcp_ops_load_hdr_opt, void *, ctx, void *, search_res, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + + /* + * No flags supported. In particular BPF_LOAD_HDR_OPT_TCP_SYN, which + * loads from the saved SYN, is not available because bpf_tcp_ops has no + * carrier to track the SYN source across the hooks. + */ + if (flags) + return -EINVAL; + + sock_ops.skb = skb; + sock_ops.skb_data_end = skb->data + tcp_hdrlen(skb); + + return __bpf_sock_ops_load_hdr_opt(&sock_ops, search_res, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_load_hdr_opt_proto = { + .func = bpf_tcp_ops_load_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_WRITE, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_3(bpf_tcp_ops_reserve_hdr_opt, void *, ctx, u32, len, u64, flags) +{ + u64 *args = ctx; + unsigned int *remaining = (void *)(unsigned long)args[5]; + + if (flags || len < 2) + return -EINVAL; + + if (len > *remaining) + return -ENOSPC; + + *remaining -= len; + return 0; +} + +static const struct bpf_func_proto bpf_tcp_ops_reserve_hdr_opt_proto = { + .func = bpf_tcp_ops_reserve_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_ANYTHING, + .arg3_type = ARG_ANYTHING, +}; + +BPF_CALL_0(bpf_tcp_ops_get_retval) +{ + struct bpf_tramp_run_ctx *ctx = + container_of(current->bpf_ctx, struct bpf_tramp_run_ctx, run_ctx); + + /* bpf_get_retval() is only exposed to timeout_init/rwnd_init, which + * always run via bpf_tcp_ops_call_int(). Its run_ctx carries the int + * return value chained across the bpf_tcp_ops attached to the cgroup + * and is this program's saved_run_ctx. + */ + if (WARN_ON_ONCE(!ctx->saved_run_ctx)) + return 0; + + return container_of(ctx->saved_run_ctx, struct bpf_tramp_run_ctx, + run_ctx)->retval; +} + +const struct bpf_func_proto bpf_tcp_ops_get_retval_proto = { + .func = bpf_tcp_ops_get_retval, + .gpl_only = false, + .ret_type = RET_INTEGER, +}; + +static const struct bpf_func_proto * +get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) +{ + u32 moff = prog->aux->attach_st_ops_member_off; + + switch (func_id) { + case BPF_FUNC_sk_storage_get: + return &bpf_sk_storage_get_proto; + case BPF_FUNC_sk_storage_delete: + return &bpf_sk_storage_delete_proto; + case BPF_FUNC_setsockopt: + /* The sk may be an unlocked listener (synack path) or NULL + * fullsock; disable for members that can run unlocked. + */ + if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return NULL; + return &bpf_sk_setsockopt_proto; + case BPF_FUNC_getsockopt: + if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return NULL; + return &bpf_sk_getsockopt_proto; + case BPF_FUNC_get_retval: + if (moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, rwnd_init)) + return &bpf_tcp_ops_get_retval_proto; + return NULL; + case BPF_FUNC_reserve_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, hdr_opt_len)) + return &bpf_tcp_ops_reserve_hdr_opt_proto; + return NULL; + case BPF_FUNC_load_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, parse_hdr) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_load_hdr_opt_proto; + return NULL; + case BPF_FUNC_store_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_store_hdr_opt_proto; + return NULL; + default: + return bpf_base_func_proto(func_id, prog); + } +} + +static bool is_valid_access(int off, int size, enum bpf_access_type type, + const struct bpf_prog *prog, struct bpf_insn_access_aux *info) +{ + if (!bpf_tracing_btf_ctx_access(off, size, type, prog, info)) + return false; + + if (base_type(info->reg_type) == PTR_TO_BTF_ID && + !bpf_type_has_unsafe_modifiers(info->reg_type) && + info->btf_id == btf_sock_ids[BTF_SOCK_TYPE_SOCK]) + /* promote it to tcp_sock */ + info->btf_id = btf_sock_ids[BTF_SOCK_TYPE_TCP]; + + return true; +} + +static int bpf_tcp_ops_init_member(const struct btf_type *t, + const struct btf_member *member, + void *kdata, const void *udata) +{ + return 0; +} + +static int bpf_tcp_ops_check_member(const struct btf_type *t, + const struct btf_member *member, + const struct bpf_prog *prog) +{ + if (prog->sleepable) + return -EINVAL; + + return 0; +} + +static int bpf_tcp_ops_init(struct btf *btf) +{ + return 0; +} + +static int bpf_tcp_ops_validate(void *kdata) +{ + return 0; +} + +static const struct bpf_verifier_ops bpf_tcp_ops_verifier = { + .get_func_proto = get_func_proto, + .is_valid_access = is_valid_access, +}; + +static struct bpf_struct_ops bpf_tcp_ops = { + .verifier_ops = &bpf_tcp_ops_verifier, + .init_member = bpf_tcp_ops_init_member, + .check_member = bpf_tcp_ops_check_member, + .init = bpf_tcp_ops_init, + .validate = bpf_tcp_ops_validate, + .name = "bpf_tcp_ops", + .cgroup_atype = CGROUP_TCP_SOCK_OPS, + .cfi_stubs = &__bpf_tcp_ops, + .owner = THIS_MODULE, +}; + +static int __init __bpf_tcp_ops_init(void) +{ + return register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops); +} +late_initcall(__bpf_tcp_ops_init); diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c index b32242a46d4a..64df53c4c133 100644 --- a/net/ipv4/tcp.c +++ b/net/ipv4/tcp.c @@ -2984,6 +2984,7 @@ void tcp_set_state(struct sock *sk, int state) if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_STATE_CB_FLAG)) tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_STATE_CB, oldstate, state); + bpf_tcp_ops_call(set_state, sk, state); switch (state) { case TCP_ESTABLISHED: diff --git a/net/ipv4/tcp_bpf.c b/net/ipv4/tcp_bpf.c index 338c5d90e550..25409563dfe9 100644 --- a/net/ipv4/tcp_bpf.c +++ b/net/ipv4/tcp_bpf.c @@ -108,7 +108,6 @@ static int tcp_bpf_push(struct sock *sk, struct sk_msg *msg, u32 apply_bytes, off = sge->offset; page = sg_page(sge); - tcp_rate_check_app_limited(sk); retry: msghdr.msg_flags = flags | MSG_SPLICE_PAGES; has_tx_ulp = tls_sw_has_ctx_tx(sk); diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c index 01d748230135..12424f45779a 100644 --- a/net/ipv4/tcp_input.c +++ b/net/ipv4/tcp_input.c @@ -208,6 +208,18 @@ static void bpf_skops_established(struct sock *sk, int bpf_op, } #endif +static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb) +{ + switch (sk->sk_state) { + case TCP_SYN_RECV: + case TCP_SYN_SENT: + case TCP_LISTEN: + return; + } + + bpf_tcp_ops_call(parse_hdr, sk, skb); +} + static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb, unsigned int len) { @@ -6474,6 +6486,7 @@ syn_challenge: pass: bpf_skops_parse_hdr(sk, skb); + bpf_tcp_ops_parse_hdr(sk, skb); return true; @@ -6738,6 +6751,10 @@ void tcp_init_transfer(struct sock *sk, int bpf_op, struct sk_buff *skb) tp->snd_cwnd_stamp = tcp_jiffies32; bpf_skops_established(sk, bpf_op, skb); + if (bpf_op == BPF_SOCK_OPS_ACTIVE_ESTABLISHED_CB) + bpf_tcp_ops_call(active_established, sk, skb); + else + bpf_tcp_ops_call(passive_established, sk, skb); /* Initialize congestion control unless BPF initialized it already: */ if (!icsk->icsk_ca_initialized) tcp_init_congestion_control(sk); diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c index c8865205cdc8..b284f6fa603a 100644 --- a/net/ipv4/tcp_output.c +++ b/net/ipv4/tcp_output.c @@ -536,43 +536,53 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, enum tcp_synack_type synack_type, struct tcp_out_options *opts) { - u8 first_opt_off, nr_written, max_opt_len = opts->bpf_opt_len; - struct bpf_sock_ops_kern sock_ops; - int err; + u8 first_opt_off, nr_written = 0, max_opt_len = opts->bpf_opt_len; if (likely(!max_opt_len)) return; - memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); + first_opt_off = tcp_hdrlen(skb) - max_opt_len; - sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; + if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), + BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG)) { + struct bpf_sock_ops_kern sock_ops; + int err; - if (req) { - sock_ops.sk = (struct sock *)req; - sock_ops.syn_skb = syn_skb; - } else { - sock_owned_by_me(sk); + memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); - sock_ops.is_fullsock = 1; - sock_ops.is_locked_tcp_sock = 1; - sock_ops.sk = sk; - } + sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; - sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); - sock_ops.remaining_opt_len = max_opt_len; - first_opt_off = tcp_hdrlen(skb) - max_opt_len; - bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + if (req) { + sock_ops.sk = (struct sock *)req; + sock_ops.syn_skb = syn_skb; + } else { + sock_owned_by_me(sk); - err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + sock_ops.is_fullsock = 1; + sock_ops.is_locked_tcp_sock = 1; + sock_ops.sk = sk; + } - if (err) - nr_written = 0; - else - nr_written = max_opt_len - sock_ops.remaining_opt_len; + sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); + sock_ops.remaining_opt_len = max_opt_len; + bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + + err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + if (!err) + nr_written = max_opt_len - sock_ops.remaining_opt_len; + } if (nr_written < max_opt_len) memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP, max_opt_len - nr_written); + + /* + * bpf_tcp_ops portion is NOP-filled (everything past the sockops + * writer's bytes). The writer finds the append point by scanning from + * first_opt_off + nr_written to the first NOP. + */ + bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type, + first_opt_off + nr_written); } #else static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, @@ -594,6 +604,32 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, } #endif +static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, + struct request_sock *req, + struct sk_buff *syn_skb, + enum tcp_synack_type synack_type, + struct tcp_out_options *opts, + u32 remaining) +{ + unsigned int remaining_out = remaining, reserved; + + if (!remaining) + return 0; + + /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */ + bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out); + + reserved = remaining - remaining_out; + if (!reserved) + return remaining; + + /* round up to 4 bytes */ + reserved = (reserved + 3) & ~3; + + opts->bpf_opt_len += reserved; + return remaining - reserved; +} + static __be32 *process_tcp_ao_options(struct tcp_sock *tp, const struct tcp_request_sock *tcprsk, struct tcp_out_options *opts, @@ -1053,6 +1089,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb, remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1141,6 +1179,8 @@ static unsigned int tcp_synack_options(const struct sock *sk, remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, synack_type, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, + synack_type, opts, remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1157,6 +1197,7 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb unsigned int eff_sacks; opts->options = 0; + opts->bpf_opt_len = 0; /* Better than switch (key.type) as it has static branches */ if (tcp_key_is_md5(key)) { @@ -1244,6 +1285,15 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb size = MAX_TCP_OPTION_SPACE - remaining; } + if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { + unsigned int remaining = MAX_TCP_OPTION_SPACE - size; + + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); + + size = MAX_TCP_OPTION_SPACE - remaining; + } + return size; } @@ -3686,6 +3736,7 @@ start: if (BPF_SOCK_OPS_TEST_FLAG(tp, BPF_SOCK_OPS_RETRANS_CB_FLAG)) tcp_call_bpf_3arg(sk, BPF_SOCK_OPS_RETRANS_CB, TCP_SKB_CB(skb)->seq, segs, err); + bpf_tcp_ops_call(retrans, sk, skb, err); if (unlikely(err) && err != -EBUSY) NET_ADD_STATS(sock_net(sk), LINUX_MIB_TCPRETRANSFAIL, segs); @@ -4313,6 +4364,7 @@ int tcp_connect(struct sock *sk) int err; tcp_call_bpf(sk, BPF_SOCK_OPS_TCP_CONNECT_CB, 0, NULL); + bpf_tcp_ops_call(connect, sk); #if defined(CONFIG_TCP_MD5SIG) && defined(CONFIG_TCP_AO) /* Has to be checked late, after setting daddr/saddr/ops. diff --git a/net/ipv4/tcp_timer.c b/net/ipv4/tcp_timer.c index e56eae4bc341..3d49adc51766 100644 --- a/net/ipv4/tcp_timer.c +++ b/net/ipv4/tcp_timer.c @@ -290,6 +290,7 @@ static int tcp_write_timeout(struct sock *sk) tcp_call_bpf_3arg(sk, BPF_SOCK_OPS_RTO_CB, icsk->icsk_retransmits, icsk->icsk_rto, (int)expired); + bpf_tcp_ops_call(rto, sk); if (expired) { /* Has it gone just too far? */ diff --git a/net/sched/bpf_qdisc.c b/net/sched/bpf_qdisc.c index 098ca02aed89..5691c13781a8 100644 --- a/net/sched/bpf_qdisc.c +++ b/net/sched/bpf_qdisc.c @@ -280,7 +280,6 @@ BTF_KFUNCS_START(qdisc_kfunc_ids) BTF_ID_FLAGS(func, bpf_skb_get_hash) BTF_ID_FLAGS(func, bpf_kfree_skb, KF_RELEASE) BTF_ID_FLAGS(func, bpf_qdisc_skb_drop, KF_RELEASE) -BTF_ID_FLAGS(func, bpf_dynptr_from_skb) BTF_ID_FLAGS(func, bpf_qdisc_watchdog_schedule) BTF_ID_FLAGS(func, bpf_qdisc_init_prologue) BTF_ID_FLAGS(func, bpf_qdisc_reset_destroy_epilogue) @@ -290,7 +289,6 @@ BTF_KFUNCS_END(qdisc_kfunc_ids) BTF_SET_START(qdisc_common_kfunc_set) BTF_ID(func, bpf_skb_get_hash) BTF_ID(func, bpf_kfree_skb) -BTF_ID(func, bpf_dynptr_from_skb) BTF_SET_END(qdisc_common_kfunc_set) BTF_SET_START(qdisc_enqueue_kfunc_set) |
