diff options
| author | Amery Hung <ameryhung@gmail.com> | 2026-09-17 13:05:38 -0700 |
|---|---|---|
| committer | Alexei Starovoitov <ast@kernel.org> | 2026-09-19 05:32:56 +0000 |
| commit | 3bb54768fe3e61f85ca7666ce119628f2e251dde (patch) | |
| tree | cba306c857aeeaa0b35e99f52f6ad5ebeb3d87f0 /net/ipv4 | |
| parent | 5ac77ae329400929f944437df6f5791467738409 (diff) | |
| download | linux-next-3bb54768fe3e61f85ca7666ce119628f2e251dde.tar.gz linux-next-3bb54768fe3e61f85ca7666ce119628f2e251dde.zip | |
bpf: tcp: Support parse/len/write header option hooks in bpf_tcp_ops
Add the TCP header option callbacks to the bpf_tcp_ops struct_ops type:
parse_hdr - parse the options of an incoming skb on an established
connection
hdr_opt_len - reserve space in the TCP header for bpf options
write_hdr_opt - write the reserved bpf options
These mirror the BPF_SOCK_OPS_PARSE_HDR_OPT_CB, _HDR_OPT_LEN_CB and
_WRITE_HDR_OPT_CB legacy sockops callbacks, but are exposed as struct_ops
members so a program can implement them with normal function signatures
and per-member helper sets.
The reserved header window is shared between the legacy sockops and
bpf_tcp_ops paths. tcp_{syn,synack,established}_options() first run the
legacy BPF_SOCK_OPS_HDR_OPT_LEN_CB and then call hdr_opt_len, so both
sources accumulate into opts->bpf_opt_len; at write time the legacy
options are emitted first and bpf_tcp_ops writes after them.
API design
bpf_tcp_ops overloads the sock_ops header-option helpers rather than
introducing a new API: bpf_reserve_hdr_opt(), bpf_store_hdr_opt() and
bpf_load_hdr_opt() are exposed per-member (reserve for hdr_opt_len,
store/load for write_hdr_opt, load for parse_hdr) and share the existing
kernel option-walking core via _bpf_sock_ops{store,load}hdr_opt(), with
the bpf_tcp_ops wrappers synthesizing a temporary bpf_sock_ops_kern from
the program ctx. This keeps a port from the legacy
BPF_SOCK_OPS*_HDR_OPT_CB callbacks mechanical (same helper calls) and
adds no new UAPI helper/kfunc surface.
An alternative considered was to drop the option helpers entirely: have
hdr_opt_len reserve space purely through its return value, and introduce
a dedicated TCP-header-option dynptr used for both reading and writing.
That is a cleaner, more self-contained interface, but it is a larger
change and does not reuse the legacy helpers, making a port from sockops
less mechanical. It can be pursued as a follow-up; the helper-based
interface here keeps this series focused on moving the hooks to
struct_ops.
The hdr_opt_len fast path in tcp_established_options() is gated by
cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS). Note this is a global,
per-attach-type static branch: it is enabled whenever any bpf_tcp_ops is
attached, even one that does not implement hdr_opt_len or that is attached
to a different cgroup. In those cases the block still runs but
bpf_tcp_ops_hdr_opt_len() no-ops via the per-member check in the dispatch
macro. A per-member/per-cgroup gate could be added later if the extra
fast-path work proves measurable.
Signed-off-by: Amery Hung <ameryhung@gmail.com>
Signed-off-by: Alexei Starovoitov <ast@kernel.org>
Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com>
Link: https://patch.msgid.link/20260917200542.3689605-13-ameryhung@gmail.com
Diffstat (limited to 'net/ipv4')
| -rw-r--r-- | net/ipv4/bpf_tcp_ops.c | 144 | ||||
| -rw-r--r-- | net/ipv4/tcp_input.c | 13 | ||||
| -rw-r--r-- | net/ipv4/tcp_output.c | 96 |
3 files changed, 227 insertions, 26 deletions
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c index 3febbc8dd1a0..681fed642999 100644 --- a/net/ipv4/bpf_tcp_ops.c +++ b/net/ipv4/bpf_tcp_ops.c @@ -4,6 +4,7 @@ #include <linux/bpf.h> #include <linux/btf_ids.h> #include <linux/bpf_verifier.h> +#include <linux/filter.h> #include <net/bpf_sk_storage.h> #include <net/tcp.h> @@ -55,6 +56,26 @@ static void listen_stub(struct sock *sk) { } +static void parse_hdr_stub(struct sock *sk, struct sk_buff *skb) +{ +} + +static void hdr_opt_len_stub(struct sock *sk, struct sk_buff *skb__nullable, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + unsigned int *remaining) +{ +} + +static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb, + struct request_sock *req__nullable, + struct sk_buff *syn_skb__nullable, + enum tcp_synack_type synack_type, + u32 opt_off) +{ +} + static struct bpf_tcp_ops __bpf_tcp_ops = { .timeout_init = timeout_init_stub, .rwnd_init = rwnd_init_stub, @@ -66,6 +87,104 @@ static struct bpf_tcp_ops __bpf_tcp_ops = { .retrans = retrans_stub, .connect = connect_stub, .listen = listen_stub, + .parse_hdr = parse_hdr_stub, + .hdr_opt_len = hdr_opt_len_stub, + .write_hdr_opt = write_hdr_opt_stub, +}; + +BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + u32 opt_off = args[5]; + u8 *op, *opend; + + /* + * bpf_tcp_ops does not keep track of the end of the written TCP header + * options, so search for it every time the helper is called. The free + * space is NOP-filled, so a TCPOPT_NOP ends the search rather than being + * skipped as in a normal option walk in sockops. + */ + op = skb->data + opt_off; + opend = skb->data + tcp_hdrlen(skb); + while (op < opend && *op != TCPOPT_NOP) { + if (*op == TCPOPT_EOL || op + 1 >= opend || op[1] < 2) + break; + op += op[1]; + } + + sock_ops.skb = skb; + sock_ops.skb_data_end = op; + sock_ops.remaining_opt_len = opend - op; + + return __bpf_sock_ops_store_hdr_opt(&sock_ops, from, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_store_hdr_opt_proto = { + .func = bpf_tcp_ops_store_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_RDONLY, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_4(bpf_tcp_ops_load_hdr_opt, void *, ctx, void *, search_res, + u32, len, u64, flags) +{ + u64 *args = ctx; + struct sk_buff *skb = (void *)(unsigned long)args[1]; + struct bpf_sock_ops_kern sock_ops = {}; + + /* + * No flags supported. In particular BPF_LOAD_HDR_OPT_TCP_SYN, which + * loads from the saved SYN, is not available because bpf_tcp_ops has no + * carrier to track the SYN source across the hooks. + */ + if (flags) + return -EINVAL; + + sock_ops.skb = skb; + sock_ops.skb_data_end = skb->data + tcp_hdrlen(skb); + + return __bpf_sock_ops_load_hdr_opt(&sock_ops, search_res, len, flags); +} + +static const struct bpf_func_proto bpf_tcp_ops_load_hdr_opt_proto = { + .func = bpf_tcp_ops_load_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_PTR_TO_MEM | MEM_WRITE, + .arg3_type = ARG_MEM_SIZE, + .arg4_type = ARG_ANYTHING, +}; + +BPF_CALL_3(bpf_tcp_ops_reserve_hdr_opt, void *, ctx, u32, len, u64, flags) +{ + u64 *args = ctx; + unsigned int *remaining = (void *)(unsigned long)args[5]; + + if (flags || len < 2) + return -EINVAL; + + if (len > *remaining) + return -ENOSPC; + + *remaining -= len; + return 0; +} + +static const struct bpf_func_proto bpf_tcp_ops_reserve_hdr_opt_proto = { + .func = bpf_tcp_ops_reserve_hdr_opt, + .gpl_only = false, + .ret_type = RET_INTEGER, + .arg1_type = ARG_PTR_TO_CTX, + .arg2_type = ARG_ANYTHING, + .arg3_type = ARG_ANYTHING, }; BPF_CALL_0(bpf_tcp_ops_get_retval) @@ -102,14 +221,20 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) case BPF_FUNC_sk_storage_delete: return &bpf_sk_storage_delete_proto; case BPF_FUNC_setsockopt: - /* The listener is not locked. */ + /* The sk may be an unlocked listener (synack path) or NULL + * fullsock; disable for members that can run unlocked. + */ if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || - moff == offsetof(struct bpf_tcp_ops, timeout_init)) + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) return NULL; return &bpf_sk_setsockopt_proto; case BPF_FUNC_getsockopt: if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) || - moff == offsetof(struct bpf_tcp_ops, timeout_init)) + moff == offsetof(struct bpf_tcp_ops, timeout_init) || + moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) return NULL; return &bpf_sk_getsockopt_proto; case BPF_FUNC_get_retval: @@ -117,6 +242,19 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) moff == offsetof(struct bpf_tcp_ops, rwnd_init)) return &bpf_tcp_ops_get_retval_proto; return NULL; + case BPF_FUNC_reserve_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, hdr_opt_len)) + return &bpf_tcp_ops_reserve_hdr_opt_proto; + return NULL; + case BPF_FUNC_load_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, parse_hdr) || + moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_load_hdr_opt_proto; + return NULL; + case BPF_FUNC_store_hdr_opt: + if (moff == offsetof(struct bpf_tcp_ops, write_hdr_opt)) + return &bpf_tcp_ops_store_hdr_opt_proto; + return NULL; default: return bpf_base_func_proto(func_id, prog); } diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c index 69e6f3925073..6ac6f9d5b6c3 100644 --- a/net/ipv4/tcp_input.c +++ b/net/ipv4/tcp_input.c @@ -208,6 +208,18 @@ static void bpf_skops_established(struct sock *sk, int bpf_op, } #endif +static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb) +{ + switch (sk->sk_state) { + case TCP_SYN_RECV: + case TCP_SYN_SENT: + case TCP_LISTEN: + return; + } + + bpf_tcp_ops_call(parse_hdr, sk, skb); +} + static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb, unsigned int len) { @@ -6461,6 +6473,7 @@ syn_challenge: pass: bpf_skops_parse_hdr(sk, skb); + bpf_tcp_ops_parse_hdr(sk, skb); return true; diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c index f75a5a01d621..908944d409d6 100644 --- a/net/ipv4/tcp_output.c +++ b/net/ipv4/tcp_output.c @@ -536,43 +536,53 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, enum tcp_synack_type synack_type, struct tcp_out_options *opts) { - u8 first_opt_off, nr_written, max_opt_len = opts->bpf_opt_len; - struct bpf_sock_ops_kern sock_ops; - int err; + u8 first_opt_off, nr_written = 0, max_opt_len = opts->bpf_opt_len; if (likely(!max_opt_len)) return; - memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); + first_opt_off = tcp_hdrlen(skb) - max_opt_len; - sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; + if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), + BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG)) { + struct bpf_sock_ops_kern sock_ops; + int err; - if (req) { - sock_ops.sk = (struct sock *)req; - sock_ops.syn_skb = syn_skb; - } else { - sock_owned_by_me(sk); + memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp)); - sock_ops.is_fullsock = 1; - sock_ops.is_locked_tcp_sock = 1; - sock_ops.sk = sk; - } + sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB; - sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); - sock_ops.remaining_opt_len = max_opt_len; - first_opt_off = tcp_hdrlen(skb) - max_opt_len; - bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + if (req) { + sock_ops.sk = (struct sock *)req; + sock_ops.syn_skb = syn_skb; + } else { + sock_owned_by_me(sk); - err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + sock_ops.is_fullsock = 1; + sock_ops.is_locked_tcp_sock = 1; + sock_ops.sk = sk; + } - if (err) - nr_written = 0; - else - nr_written = max_opt_len - sock_ops.remaining_opt_len; + sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type); + sock_ops.remaining_opt_len = max_opt_len; + bpf_skops_init_skb(&sock_ops, skb, first_opt_off); + + err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk); + if (!err) + nr_written = max_opt_len - sock_ops.remaining_opt_len; + } if (nr_written < max_opt_len) memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP, max_opt_len - nr_written); + + /* + * bpf_tcp_ops portion is NOP-filled (everything past the sockops + * writer's bytes). The writer finds the append point by scanning from + * first_opt_off + nr_written to the first NOP. + */ + bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type, + first_opt_off + nr_written); } #else static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, @@ -594,6 +604,32 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, } #endif +static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, + struct request_sock *req, + struct sk_buff *syn_skb, + enum tcp_synack_type synack_type, + struct tcp_out_options *opts, + u32 remaining) +{ + unsigned int remaining_out = remaining, reserved; + + if (!remaining) + return 0; + + /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */ + bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out); + + reserved = remaining - remaining_out; + if (!reserved) + return remaining; + + /* round up to 4 bytes */ + reserved = (reserved + 3) & ~3; + + opts->bpf_opt_len += reserved; + return remaining - reserved; +} + static __be32 *process_tcp_ao_options(struct tcp_sock *tp, const struct tcp_request_sock *tcprsk, struct tcp_out_options *opts, @@ -1053,6 +1089,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb, remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1141,6 +1179,8 @@ static unsigned int tcp_synack_options(const struct sock *sk, remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, synack_type, opts, remaining); + remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb, + synack_type, opts, remaining); return MAX_TCP_OPTION_SPACE - remaining; } @@ -1157,6 +1197,7 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb unsigned int eff_sacks; opts->options = 0; + opts->bpf_opt_len = 0; /* Better than switch (key.type) as it has static branches */ if (tcp_key_is_md5(key)) { @@ -1244,6 +1285,15 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb size = MAX_TCP_OPTION_SPACE - remaining; } + if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { + unsigned int remaining = MAX_TCP_OPTION_SPACE - size; + + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); + + size = MAX_TCP_OPTION_SPACE - remaining; + } + return size; } |
