diff options
| -rw-r--r-- | include/linux/filter.h | 5 | ||||
| -rw-r--r-- | include/net/tcp.h | 43 | ||||
| -rw-r--r-- | include/uapi/linux/bpf.h | 35 | ||||
| -rw-r--r-- | net/core/filter.c | 32 | ||||
| -rw-r--r-- | net/ipv4/bpf_tcp_ops.c | 144 | ||||
| -rw-r--r-- | net/ipv4/tcp_input.c | 13 | ||||
| -rw-r--r-- | net/ipv4/tcp_output.c | 96 | ||||
| -rw-r--r-- | tools/include/uapi/linux/bpf.h | 35 |
8 files changed, 341 insertions, 62 deletions
diff --git a/include/linux/filter.h b/include/linux/filter.h index b17222db2efc..422284b4fa96 100644 --- a/include/linux/filter.h +++ b/include/linux/filter.h @@ -1930,6 +1930,11 @@ static __always_inline long __bpf_xdp_redirect_map(struct bpf_map *map, u64 inde return XDP_REDIRECT; } +int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + void *search_res, u32 len, u64 flags); +int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + const void *from, u32 len, u64 flags); + #ifdef CONFIG_NET int __bpf_skb_load_bytes(const struct sk_buff *skb, u32 offset, void *to, u32 len); int __bpf_skb_store_bytes(struct sk_buff *skb, u32 offset, const void *from, diff --git a/include/net/tcp.h b/include/net/tcp.h index af3747043778..d61ee00052e3 100644 --- a/include/net/tcp.h +++ b/include/net/tcp.h @@ -3011,6 +3011,48 @@ struct bpf_tcp_ops { /* Called on listen(2), right after the socket enters TCP_LISTEN. */ void (*listen)(struct sock *sk); + + /* + * Parse the TCP header options of an incoming skb received on an + * established connection. Use bpf_dynptr_from_skb()/bpf_skb_load_bytes() + * to access the options. + */ + void (*parse_hdr)(struct sock *sk, struct sk_buff *skb); + + /* + * Reserve space in the outgoing TCP header for options to be written + * later by write_hdr_opt(). Call bpf_reserve_hdr_opt() to reserve bytes. + * + * @skb: outgoing packet. NULL when called from tcp_current_mss() + * (MSS sizing). + * @req: request_sock on the synack path; NULL otherwise. + * @syn_skb: incoming SYN on the synack path; NULL otherwise. + * @synack_type: TCP_SYNACK_COOKIE indicates a stateless syncookie. + * @remaining: pointer to the size of space still available; cast it + * using bpf_rdonly_cast() before dereferencing. + */ + void (*hdr_opt_len)(struct sock *sk, struct sk_buff *skb, + struct request_sock *req, struct sk_buff *syn_skb, + enum tcp_synack_type synack_type, + unsigned int *remaining); + + /* + * Write header options into the space reserved earlier by hdr_opt_len(). + * Use bpf_store_hdr_opt() to write; it appends within the reserved window + * shared with legacy SOCKOPS. + * + * @skb: outgoing packet. + * @req: request_sock on the synack path; NULL otherwise. + * @syn_skb: incoming SYN on the synack path; NULL otherwise. + * @synack_type: TCP_SYNACK_COOKIE indicates a stateless syncookie. + * @opt_off: offset in the outgoing @skb's TCP header where the + * bpf_tcp_ops portion of the reserved window begins, i.e. after + * the kernel and legacy options. + */ + void (*write_hdr_opt)(struct sock *sk, struct sk_buff *skb, + struct request_sock *req, struct sk_buff *syn_skb, + enum tcp_synack_type synack_type, + u32 opt_off); }; #define bpf_tcp_ops_call(op, sk, ...) \ @@ -3062,6 +3104,7 @@ do { \ } \ __retval; \ }) + #else #define bpf_tcp_ops_call(op, sk, ...) do { } while (0) #define bpf_tcp_ops_call_int(op, init_retval, sk, ...) (init_retval) diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h index dabe01cd3def..6330b7d745c5 100644 --- a/include/uapi/linux/bpf.h +++ b/include/uapi/linux/bpf.h @@ -4867,15 +4867,18 @@ union bpf_attr { * The non-negative copied *buf* length equal to or less than * *size* on success, or a negative error in case of failure. * - * long bpf_load_hdr_opt(struct bpf_sock_ops *skops, void *searchby_res, u32 len, u64 flags) + * long bpf_load_hdr_opt(void *ctx, void *searchby_res, u32 len, u64 flags) * Description * Load header option. Support reading a particular TCP header - * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). + * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). For the + * **bpf_tcp_ops** struct_ops, this helper can be called from the + * **parse_hdr**\ () and **write_hdr_opt**\ () operators. * - * If *flags* is 0, it will search the option from the - * *skops*\ **->skb_data**. The comment in **struct bpf_sock_ops** - * has details on what skb_data contains under different - * *skops*\ **->op**. + * If *flags* is 0, it will search the option from the packet + * associated with the current operation. For + * **BPF_PROG_TYPE_SOCK_OPS**, the comment in + * **struct bpf_sock_ops** has details on what skb_data + * contains under different *op*. * * The first byte of the *searchby_res* specifies the * kind that it wants to search. @@ -4908,6 +4911,8 @@ union bpf_attr { * * * **BPF_LOAD_HDR_OPT_TCP_SYN** to search from the * saved_syn packet or the just-received syn packet. + * Not supported by the **bpf_tcp_ops** struct_ops, which + * rejects all flags. * * Return * > 0 when found, the header option is copied to *searchby_res*. @@ -4928,9 +4933,9 @@ union bpf_attr { * packet. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * - * long bpf_store_hdr_opt(struct bpf_sock_ops *skops, const void *from, u32 len, u64 flags) + * long bpf_store_hdr_opt(void *ctx, const void *from, u32 len, u64 flags) * Description * Store header option. The data will be copied * from buffer *from* with length *len* to the TCP header. @@ -4946,7 +4951,9 @@ union bpf_attr { * by searching the same option in the outgoing skb. * * This helper can only be called during - * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**. + * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**, or from the + * **write_hdr_opt**\ () operator of the **bpf_tcp_ops** + * struct_ops. * * Return * 0 on success, or negative error in case of failure: @@ -4961,9 +4968,9 @@ union bpf_attr { * **-EFAULT** on failure to parse the existing header options. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * - * long bpf_reserve_hdr_opt(struct bpf_sock_ops *skops, u32 len, u64 flags) + * long bpf_reserve_hdr_opt(void *ctx, u32 len, u64 flags) * Description * Reserve *len* bytes for the bpf header option. The * space will be used by **bpf_store_hdr_opt**\ () later in @@ -4973,7 +4980,9 @@ union bpf_attr { * the total number of bytes will be reserved. * * This helper can only be called during - * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**. + * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**, or from the + * **hdr_opt_len**\ () operator of the **bpf_tcp_ops** + * struct_ops. * * Return * 0 on success, or negative error in case of failure: @@ -4983,7 +4992,7 @@ union bpf_attr { * **-ENOSPC** if there is not enough space in the header. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * * void *bpf_inode_storage_get(struct bpf_map *map, void *inode, void *value, u64 flags) * Description diff --git a/net/core/filter.c b/net/core/filter.c index 47b7a8f73069..5feb99884682 100644 --- a/net/core/filter.c +++ b/net/core/filter.c @@ -8053,17 +8053,14 @@ static const u8 *bpf_search_tcp_opt(const u8 *op, const u8 *opend, return ERR_PTR(-ENOMSG); } -BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, - void *, search_res, u32, len, u64, flags) +int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + void *search_res, u32 len, u64 flags) { bool eol, load_syn = flags & BPF_LOAD_HDR_OPT_TCP_SYN; const u8 *op, *opend, *magic, *search = search_res; u8 search_kind, search_len, copy_len, magic_len; int ret; - if (!is_locked_tcp_sock_ops(bpf_sock)) - return -EOPNOTSUPP; - /* 2 byte is the minimal option len except TCPOPT_NOP and * TCPOPT_EOL which are useless for the bpf prog to learn * and this helper disallow loading them also. @@ -8124,6 +8121,15 @@ BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, return ret; } +BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, + void *, search_res, u32, len, u64, flags) +{ + if (!is_locked_tcp_sock_ops(bpf_sock)) + return -EOPNOTSUPP; + + return __bpf_sock_ops_load_hdr_opt(bpf_sock, search_res, len, flags); +} + static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = { .func = bpf_sock_ops_load_hdr_opt, .gpl_only = false, @@ -8134,17 +8140,14 @@ static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = { .arg4_type = ARG_ANYTHING, }; -BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, - const void *, from, u32, len, u64, flags) +int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock, + const void *from, u32 len, u64 flags) { u8 new_kind, new_kind_len, magic_len = 0, *opend; const u8 *op, *new_op, *magic = NULL; struct sk_buff *skb; bool eol; - if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB) - return -EPERM; - if (len < 2 || flags) return -EINVAL; @@ -8202,6 +8205,15 @@ BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, return 0; } +BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock, + const void *, from, u32, len, u64, flags) +{ + if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB) + return -EPERM; + + return __bpf_sock_ops_store_hdr_opt(bpf_sock, from, len, flags); +} + static const struct bpf_func_proto bpf_sock_ops_store_hdr_opt_proto = { .func = bpf_sock_ops_store_hdr_opt, .gpl_only = false, diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c index 3febbc8dd1a0..681fed642999 100644 --- a/net/ipv4/bpf_tcp_ops.c +++ b/net/ipv4/bpf_tcp_ops.c @@ -4,6 +4,7 @@ #include <linux/bpf.h> #include <linux/btf_ids.h> #include <linux/bpf_verifier.h> +#include <linux/filter.h> #include <net/bpf_sk_storage.h> #include <net/tcp.h> @@ -55,6 +56,26 @@ static void listen_stub(struct sock *sk) { } +static void parse_hdr_stub(struct sock *sk, struct sk_buff *skb) +{ +} + +static void hdr_opt_len_stub(struct sock *sk, struct sk_buff *skb__nullable, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + unsigned int *remaining) +{ +} + +static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + u32 opt_off) +{ +} + static struct bpf_tcp_ops __bpf_tcp_ops = { .timeout_init = timeout_init_stub, .rwnd_init = rwnd_init_stub, @@ -66,6 +87,104 @@ static struct bpf_tcp_ops __bpf_tcp_ops = { .retrans = retrans_stub, .connect = connect_stub, .listen = listen_stub, + .parse_hdr = parse_hdr_stub, + .hdr_opt_len = hdr_opt_len_stub, + .write_hdr_opt = write_hdr_opt_stub, +}; + +BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + u32 opt_off = args[5]; + u8 *op, *opend; + + /* + * bpf_tcp_ops does not keep track of the end of the written TCP header + * options, so search for it every time the helper is called. The free + * space is NOP-filled, so a TCPOPT_NOP ends the search rather than being + * skipped as in a normal option walk in sockops. + */ + op = skb->data + opt_off; + opend = skb->data + tcp_hdrlen(skb); + while (op < opend && *op != TCPOPT_NOP) { + if (*op == TCPOPT_EOL || op + 1 >= opend || op[1] < 2) + break; + op += op[1]; + } + + sock_ops.skb = skb; + sock_ops.skb_data_end = op; + sock_ops.remaining_opt_len = opend - op; + + return __bpf_sock_ops_store_hdr_opt(&sock_ops, from, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_store_hdr_opt_proto = { + .func = bpf_tcp_ops_store_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_RDONLY, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_4(bpf_tcp_ops_load_hdr_opt, void *, ctx, void *, search_res, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + + /* + * No flags supported. In particular BPF_LOAD_HDR_OPT_TCP_SYN, which + * loads from the saved SYN, is not available because bpf_tcp_ops has no + * carrier to track the SYN source across the hooks. + */ + if (flags) + return -EINVAL; + + sock_ops.skb = skb; + sock_ops.skb_data_end = skb->data + tcp_hdrlen(skb); + + return __bpf_sock_ops_load_hdr_opt(&sock_ops, search_res, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_load_hdr_opt_proto = { + .func = bpf_tcp_ops_load_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_WRITE, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_3(bpf_tcp_ops_reserve_hdr_opt, void *, ctx, u32, len, u64, flags) +{ + u64 *args = ctx; + unsigned int *remaining = (void *)(unsigned long)args[5]; + + if (flags || len < 2) + return -EINVAL; + + if (len > *remaining) + return -ENOSPC; + + *remaining -= len; + return 0; +} + +static const struct bpf_func_proto bpf_tcp_ops_reserve_hdr_opt_proto = { + .func = bpf_tcp_ops_reserve_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_ANYTHING, + .arg3_type = ARG_ANYTHING, }; BPF_CALL_0(bpf_tcp_ops_get_retval) @@ -102,14 +221,20 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) case BPF_FUNC_sk_storage_delete: return &bpf_sk_storage_delete_proto; case BPF_FUNC_setsockopt: - /* The listener is not locked. */ + /* The sk may be an unlocked listener (synack path) or NULL + * fullsock; disable for members that can run unlocked. + */ if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || - moff == offsetof(struct bpf_tcp_ops, timeout_init)) + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) return NULL; return &bpf_sk_setsockopt_proto; case BPF_FUNC_getsockopt: if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || - moff == offsetof(struct bpf_tcp_ops, timeout_init)) + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) return NULL; return &bpf_sk_getsockopt_proto; case BPF_FUNC_get_retval: @@ -117,6 +242,19 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) moff == offsetof(struct bpf_tcp_ops, rwnd_init)) return &bpf_tcp_ops_get_retval_proto; return NULL; + case BPF_FUNC_reserve_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, hdr_opt_len)) + return &bpf_tcp_ops_reserve_hdr_opt_proto; + return NULL; + case BPF_FUNC_load_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, parse_hdr) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_load_hdr_opt_proto; + return NULL; + case BPF_FUNC_store_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_store_hdr_opt_proto; + return NULL; default: return bpf_base_func_proto(func_id, prog); } diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c index 69e6f3925073..6ac6f9d5b6c3 100644 --- a/net/ipv4/tcp_input.c +++ b/net/ipv4/tcp_input.c @@ -208,6 +208,18 @@ static void bpf_skops_established(struct sock *sk, int bpf_op, } #endif +static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb) +{ + switch (sk->sk_state) { + case TCP_SYN_RECV: + case TCP_SYN_SENT: + case TCP_LISTEN: + return; + } + + bpf_tcp_ops_call(parse_hdr, sk, skb); +} + static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb, unsigned int len) { @@ -6461,6 +6473,7 @@ syn_challenge: pass: bpf_skops_parse_hdr(sk, skb); + bpf_tcp_ops_parse_hdr(sk, skb); return true; diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c index f75a5a01d621..908944d409d6 100644 --- a/net/ipv4/tcp_output.c +++ b/net/ipv4/tcp_output.c @@ -536,43 +536,53 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, enum tcp_synack_type synack_type, struct tcp_out_options *opts) { - u8 first_opt_off, nr_written, max_opt_len = opts->bpf_opt_len; - struct bpf_sock_ops_kern sock_ops; - int err; + u8 first_opt_off, nr_written = 0, max_opt_len = opts->bpf_opt_len; if (likely(!max_opt_len)) return; - memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); + first_opt_off = tcp_hdrlen(skb) - max_opt_len; - sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; + if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), + BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG)) { + struct bpf_sock_ops_kern sock_ops; + int err; - if (req) { - sock_ops.sk = (struct sock *)req; - sock_ops.syn_skb = syn_skb; - } else { - sock_owned_by_me(sk); + memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); - sock_ops.is_fullsock = 1; - sock_ops.is_locked_tcp_sock = 1; - sock_ops.sk = sk; - } + sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; - sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); - sock_ops.remaining_opt_len = max_opt_len; - first_opt_off = tcp_hdrlen(skb) - max_opt_len; - bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + if (req) { + sock_ops.sk = (struct sock *)req; + sock_ops.syn_skb = syn_skb; + } else { + sock_owned_by_me(sk); - err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + sock_ops.is_fullsock = 1; + sock_ops.is_locked_tcp_sock = 1; + sock_ops.sk = sk; + } - if (err) - nr_written = 0; - else - nr_written = max_opt_len - sock_ops.remaining_opt_len; + sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); + sock_ops.remaining_opt_len = max_opt_len; + bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + + err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + if (!err) + nr_written = max_opt_len - sock_ops.remaining_opt_len; + } if (nr_written < max_opt_len) memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP, max_opt_len - nr_written); + + /* + * bpf_tcp_ops portion is NOP-filled (everything past the sockops + * writer's bytes). The writer finds the append point by scanning from + * first_opt_off + nr_written to the first NOP. + */ + bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type, + first_opt_off + nr_written); } #else static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, @@ -594,6 +604,32 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, } #endif +static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, + struct request_sock *req, + struct sk_buff *syn_skb, + enum tcp_synack_type synack_type, + struct tcp_out_options *opts, + u32 remaining) +{ + unsigned int remaining_out = remaining, reserved; + + if (!remaining) + return 0; + + /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */ + bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out); + + reserved = remaining - remaining_out; + if (!reserved) + return remaining; + + /* round up to 4 bytes */ + reserved = (reserved + 3) & ~3; + + opts->bpf_opt_len += reserved; + return remaining - reserved; +} + static __be32 *process_tcp_ao_options(struct tcp_sock *tp, const struct tcp_request_sock *tcprsk, struct tcp_out_options *opts, @@ -1053,6 +1089,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb, remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1141,6 +1179,8 @@ static unsigned int tcp_synack_options(const struct sock *sk, remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, synack_type, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, + synack_type, opts, remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1157,6 +1197,7 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb unsigned int eff_sacks; opts->options = 0; + opts->bpf_opt_len = 0; /* Better than switch (key.type) as it has static branches */ if (tcp_key_is_md5(key)) { @@ -1244,6 +1285,15 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb size = MAX_TCP_OPTION_SPACE - remaining; } + if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { + unsigned int remaining = MAX_TCP_OPTION_SPACE - size; + + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); + + size = MAX_TCP_OPTION_SPACE - remaining; + } + return size; } diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h index dabe01cd3def..6330b7d745c5 100644 --- a/tools/include/uapi/linux/bpf.h +++ b/tools/include/uapi/linux/bpf.h @@ -4867,15 +4867,18 @@ union bpf_attr { * The non-negative copied *buf* length equal to or less than * *size* on success, or a negative error in case of failure. * - * long bpf_load_hdr_opt(struct bpf_sock_ops *skops, void *searchby_res, u32 len, u64 flags) + * long bpf_load_hdr_opt(void *ctx, void *searchby_res, u32 len, u64 flags) * Description * Load header option. Support reading a particular TCP header - * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). + * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). For the + * **bpf_tcp_ops** struct_ops, this helper can be called from the + * **parse_hdr**\ () and **write_hdr_opt**\ () operators. * - * If *flags* is 0, it will search the option from the - * *skops*\ **->skb_data**. The comment in **struct bpf_sock_ops** - * has details on what skb_data contains under different - * *skops*\ **->op**. + * If *flags* is 0, it will search the option from the packet + * associated with the current operation. For + * **BPF_PROG_TYPE_SOCK_OPS**, the comment in + * **struct bpf_sock_ops** has details on what skb_data + * contains under different *op*. * * The first byte of the *searchby_res* specifies the * kind that it wants to search. @@ -4908,6 +4911,8 @@ union bpf_attr { * * * **BPF_LOAD_HDR_OPT_TCP_SYN** to search from the * saved_syn packet or the just-received syn packet. + * Not supported by the **bpf_tcp_ops** struct_ops, which + * rejects all flags. * * Return * > 0 when found, the header option is copied to *searchby_res*. @@ -4928,9 +4933,9 @@ union bpf_attr { * packet. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * - * long bpf_store_hdr_opt(struct bpf_sock_ops *skops, const void *from, u32 len, u64 flags) + * long bpf_store_hdr_opt(void *ctx, const void *from, u32 len, u64 flags) * Description * Store header option. The data will be copied * from buffer *from* with length *len* to the TCP header. @@ -4946,7 +4951,9 @@ union bpf_attr { * by searching the same option in the outgoing skb. * * This helper can only be called during - * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**. + * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**, or from the + * **write_hdr_opt**\ () operator of the **bpf_tcp_ops** + * struct_ops. * * Return * 0 on success, or negative error in case of failure: @@ -4961,9 +4968,9 @@ union bpf_attr { * **-EFAULT** on failure to parse the existing header options. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * - * long bpf_reserve_hdr_opt(struct bpf_sock_ops *skops, u32 len, u64 flags) + * long bpf_reserve_hdr_opt(void *ctx, u32 len, u64 flags) * Description * Reserve *len* bytes for the bpf header option. The * space will be used by **bpf_store_hdr_opt**\ () later in @@ -4973,7 +4980,9 @@ union bpf_attr { * the total number of bytes will be reserved. * * This helper can only be called during - * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**. + * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**, or from the + * **hdr_opt_len**\ () operator of the **bpf_tcp_ops** + * struct_ops. * * Return * 0 on success, or negative error in case of failure: @@ -4983,7 +4992,7 @@ union bpf_attr { * **-ENOSPC** if there is not enough space in the header. * * **-EPERM** if the helper cannot be used under the current - * *skops*\ **->op**. + * operation. * * void *bpf_inode_storage_get(struct bpf_map *map, void *inode, void *value, u64 flags) * Description |
