summaryrefslogtreecommitdiff
path: root/net/ipv4
diff options
context:
space:
mode:
authorAmery Hung <ameryhung@gmail.com>2026-09-17 13:05:38 -0700
committerAlexei Starovoitov <ast@kernel.org>2026-09-19 05:32:56 +0000
commit3bb54768fe3e61f85ca7666ce119628f2e251dde (patch)
treecba306c857aeeaa0b35e99f52f6ad5ebeb3d87f0 /net/ipv4
parent5ac77ae329400929f944437df6f5791467738409 (diff)
downloadlinux-next-3bb54768fe3e61f85ca7666ce119628f2e251dde.tar.gz
linux-next-3bb54768fe3e61f85ca7666ce119628f2e251dde.zip
bpf: tcp: Support parse/len/write header option hooks in bpf_tcp_ops
Add the TCP header option callbacks to the bpf_tcp_ops struct_ops type: parse_hdr - parse the options of an incoming skb on an established connection hdr_opt_len - reserve space in the TCP header for bpf options write_hdr_opt - write the reserved bpf options These mirror the BPF_SOCK_OPS_PARSE_HDR_OPT_CB, _HDR_OPT_LEN_CB and _WRITE_HDR_OPT_CB legacy sockops callbacks, but are exposed as struct_ops members so a program can implement them with normal function signatures and per-member helper sets. The reserved header window is shared between the legacy sockops and bpf_tcp_ops paths. tcp_{syn,synack,established}_options() first run the legacy BPF_SOCK_OPS_HDR_OPT_LEN_CB and then call hdr_opt_len, so both sources accumulate into opts->bpf_opt_len; at write time the legacy options are emitted first and bpf_tcp_ops writes after them. API design bpf_tcp_ops overloads the sock_ops header-option helpers rather than introducing a new API: bpf_reserve_hdr_opt(), bpf_store_hdr_opt() and bpf_load_hdr_opt() are exposed per-member (reserve for hdr_opt_len, store/load for write_hdr_opt, load for parse_hdr) and share the existing kernel option-walking core via _bpf_sock_ops{store,load}hdr_opt(), with the bpf_tcp_ops wrappers synthesizing a temporary bpf_sock_ops_kern from the program ctx. This keeps a port from the legacy BPF_SOCK_OPS*_HDR_OPT_CB callbacks mechanical (same helper calls) and adds no new UAPI helper/kfunc surface. An alternative considered was to drop the option helpers entirely: have hdr_opt_len reserve space purely through its return value, and introduce a dedicated TCP-header-option dynptr used for both reading and writing. That is a cleaner, more self-contained interface, but it is a larger change and does not reuse the legacy helpers, making a port from sockops less mechanical. It can be pursued as a follow-up; the helper-based interface here keeps this series focused on moving the hooks to struct_ops. The hdr_opt_len fast path in tcp_established_options() is gated by cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS). Note this is a global, per-attach-type static branch: it is enabled whenever any bpf_tcp_ops is attached, even one that does not implement hdr_opt_len or that is attached to a different cgroup. In those cases the block still runs but bpf_tcp_ops_hdr_opt_len() no-ops via the per-member check in the dispatch macro. A per-member/per-cgroup gate could be added later if the extra fast-path work proves measurable. Signed-off-by: Amery Hung <ameryhung@gmail.com> Signed-off-by: Alexei Starovoitov <ast@kernel.org> Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com> Link: https://patch.msgid.link/20260917200542.3689605-13-ameryhung@gmail.com
Diffstat (limited to 'net/ipv4')
-rw-r--r--net/ipv4/bpf_tcp_ops.c144
-rw-r--r--net/ipv4/tcp_input.c13
-rw-r--r--net/ipv4/tcp_output.c96
3 files changed, 227 insertions, 26 deletions
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index 3febbc8dd1a0..681fed642999 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -4,6 +4,7 @@
#include <linux/bpf.h>
#include <linux/btf_ids.h>
#include <linux/bpf_verifier.h>
+#include <linux/filter.h>
#include <net/bpf_sk_storage.h>
#include <net/tcp.h>
@@ -55,6 +56,26 @@ static void listen_stub(struct sock *sk)
{
}
+static void parse_hdr_stub(struct sock *sk, struct sk_buff *skb)
+{
+}
+
+static void hdr_opt_len_stub(struct sock *sk, struct sk_buff *skb__nullable,
+ struct request_sock *req__nullable,
+ struct sk_buff *syn_skb__nullable,
+ enum tcp_synack_type synack_type,
+ unsigned int *remaining)
+{
+}
+
+static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req__nullable,
+ struct sk_buff *syn_skb__nullable,
+ enum tcp_synack_type synack_type,
+ u32 opt_off)
+{
+}
+
static struct bpf_tcp_ops __bpf_tcp_ops = {
.timeout_init = timeout_init_stub,
.rwnd_init = rwnd_init_stub,
@@ -66,6 +87,104 @@ static struct bpf_tcp_ops __bpf_tcp_ops = {
.retrans = retrans_stub,
.connect = connect_stub,
.listen = listen_stub,
+ .parse_hdr = parse_hdr_stub,
+ .hdr_opt_len = hdr_opt_len_stub,
+ .write_hdr_opt = write_hdr_opt_stub,
+};
+
+BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from,
+ u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ struct sk_buff *skb = (void *)(unsigned long)args[1];
+ struct bpf_sock_ops_kern sock_ops = {};
+ u32 opt_off = args[5];
+ u8 *op, *opend;
+
+ /*
+ * bpf_tcp_ops does not keep track of the end of the written TCP header
+ * options, so search for it every time the helper is called. The free
+ * space is NOP-filled, so a TCPOPT_NOP ends the search rather than being
+ * skipped as in a normal option walk in sockops.
+ */
+ op = skb->data + opt_off;
+ opend = skb->data + tcp_hdrlen(skb);
+ while (op < opend && *op != TCPOPT_NOP) {
+ if (*op == TCPOPT_EOL || op + 1 >= opend || op[1] < 2)
+ break;
+ op += op[1];
+ }
+
+ sock_ops.skb = skb;
+ sock_ops.skb_data_end = op;
+ sock_ops.remaining_opt_len = opend - op;
+
+ return __bpf_sock_ops_store_hdr_opt(&sock_ops, from, len, flags);
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_store_hdr_opt_proto = {
+ .func = bpf_tcp_ops_store_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_PTR_TO_MEM | MEM_RDONLY,
+ .arg3_type = ARG_MEM_SIZE,
+ .arg4_type = ARG_ANYTHING,
+};
+
+BPF_CALL_4(bpf_tcp_ops_load_hdr_opt, void *, ctx, void *, search_res,
+ u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ struct sk_buff *skb = (void *)(unsigned long)args[1];
+ struct bpf_sock_ops_kern sock_ops = {};
+
+ /*
+ * No flags supported. In particular BPF_LOAD_HDR_OPT_TCP_SYN, which
+ * loads from the saved SYN, is not available because bpf_tcp_ops has no
+ * carrier to track the SYN source across the hooks.
+ */
+ if (flags)
+ return -EINVAL;
+
+ sock_ops.skb = skb;
+ sock_ops.skb_data_end = skb->data + tcp_hdrlen(skb);
+
+ return __bpf_sock_ops_load_hdr_opt(&sock_ops, search_res, len, flags);
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_load_hdr_opt_proto = {
+ .func = bpf_tcp_ops_load_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_PTR_TO_MEM | MEM_WRITE,
+ .arg3_type = ARG_MEM_SIZE,
+ .arg4_type = ARG_ANYTHING,
+};
+
+BPF_CALL_3(bpf_tcp_ops_reserve_hdr_opt, void *, ctx, u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ unsigned int *remaining = (void *)(unsigned long)args[5];
+
+ if (flags || len < 2)
+ return -EINVAL;
+
+ if (len > *remaining)
+ return -ENOSPC;
+
+ *remaining -= len;
+ return 0;
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_reserve_hdr_opt_proto = {
+ .func = bpf_tcp_ops_reserve_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_ANYTHING,
+ .arg3_type = ARG_ANYTHING,
};
BPF_CALL_0(bpf_tcp_ops_get_retval)
@@ -102,14 +221,20 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
case BPF_FUNC_sk_storage_delete:
return &bpf_sk_storage_delete_proto;
case BPF_FUNC_setsockopt:
- /* The listener is not locked. */
+ /* The sk may be an unlocked listener (synack path) or NULL
+ * fullsock; disable for members that can run unlocked.
+ */
if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init))
+ moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
+ moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
return NULL;
return &bpf_sk_setsockopt_proto;
case BPF_FUNC_getsockopt:
if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init))
+ moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
+ moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
return NULL;
return &bpf_sk_getsockopt_proto;
case BPF_FUNC_get_retval:
@@ -117,6 +242,19 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
moff == offsetof(struct bpf_tcp_ops, rwnd_init))
return &bpf_tcp_ops_get_retval_proto;
return NULL;
+ case BPF_FUNC_reserve_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, hdr_opt_len))
+ return &bpf_tcp_ops_reserve_hdr_opt_proto;
+ return NULL;
+ case BPF_FUNC_load_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, parse_hdr) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
+ return &bpf_tcp_ops_load_hdr_opt_proto;
+ return NULL;
+ case BPF_FUNC_store_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
+ return &bpf_tcp_ops_store_hdr_opt_proto;
+ return NULL;
default:
return bpf_base_func_proto(func_id, prog);
}
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 69e6f3925073..6ac6f9d5b6c3 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -208,6 +208,18 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
}
#endif
+static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
+{
+ switch (sk->sk_state) {
+ case TCP_SYN_RECV:
+ case TCP_SYN_SENT:
+ case TCP_LISTEN:
+ return;
+ }
+
+ bpf_tcp_ops_call(parse_hdr, sk, skb);
+}
+
static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb,
unsigned int len)
{
@@ -6461,6 +6473,7 @@ syn_challenge:
pass:
bpf_skops_parse_hdr(sk, skb);
+ bpf_tcp_ops_parse_hdr(sk, skb);
return true;
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index f75a5a01d621..908944d409d6 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -536,43 +536,53 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
enum tcp_synack_type synack_type,
struct tcp_out_options *opts)
{
- u8 first_opt_off, nr_written, max_opt_len = opts->bpf_opt_len;
- struct bpf_sock_ops_kern sock_ops;
- int err;
+ u8 first_opt_off, nr_written = 0, max_opt_len = opts->bpf_opt_len;
if (likely(!max_opt_len))
return;
- memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp));
+ first_opt_off = tcp_hdrlen(skb) - max_opt_len;
- sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB;
+ if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk),
+ BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG)) {
+ struct bpf_sock_ops_kern sock_ops;
+ int err;
- if (req) {
- sock_ops.sk = (struct sock *)req;
- sock_ops.syn_skb = syn_skb;
- } else {
- sock_owned_by_me(sk);
+ memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp));
- sock_ops.is_fullsock = 1;
- sock_ops.is_locked_tcp_sock = 1;
- sock_ops.sk = sk;
- }
+ sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB;
- sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type);
- sock_ops.remaining_opt_len = max_opt_len;
- first_opt_off = tcp_hdrlen(skb) - max_opt_len;
- bpf_skops_init_skb(&sock_ops, skb, first_opt_off);
+ if (req) {
+ sock_ops.sk = (struct sock *)req;
+ sock_ops.syn_skb = syn_skb;
+ } else {
+ sock_owned_by_me(sk);
- err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk);
+ sock_ops.is_fullsock = 1;
+ sock_ops.is_locked_tcp_sock = 1;
+ sock_ops.sk = sk;
+ }
- if (err)
- nr_written = 0;
- else
- nr_written = max_opt_len - sock_ops.remaining_opt_len;
+ sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type);
+ sock_ops.remaining_opt_len = max_opt_len;
+ bpf_skops_init_skb(&sock_ops, skb, first_opt_off);
+
+ err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk);
+ if (!err)
+ nr_written = max_opt_len - sock_ops.remaining_opt_len;
+ }
if (nr_written < max_opt_len)
memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP,
max_opt_len - nr_written);
+
+ /*
+ * bpf_tcp_ops portion is NOP-filled (everything past the sockops
+ * writer's bytes). The writer finds the append point by scanning from
+ * first_opt_off + nr_written to the first NOP.
+ */
+ bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
+ first_opt_off + nr_written);
}
#else
static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -594,6 +604,32 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
}
#endif
+static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req,
+ struct sk_buff *syn_skb,
+ enum tcp_synack_type synack_type,
+ struct tcp_out_options *opts,
+ u32 remaining)
+{
+ unsigned int remaining_out = remaining, reserved;
+
+ if (!remaining)
+ return 0;
+
+ /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
+ bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
+
+ reserved = remaining - remaining_out;
+ if (!reserved)
+ return remaining;
+
+ /* round up to 4 bytes */
+ reserved = (reserved + 3) & ~3;
+
+ opts->bpf_opt_len += reserved;
+ return remaining - reserved;
+}
+
static __be32 *process_tcp_ao_options(struct tcp_sock *tp,
const struct tcp_request_sock *tcprsk,
struct tcp_out_options *opts,
@@ -1053,6 +1089,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb,
remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
remaining);
+ remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+ remaining);
return MAX_TCP_OPTION_SPACE - remaining;
}
@@ -1141,6 +1179,8 @@ static unsigned int tcp_synack_options(const struct sock *sk,
remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
synack_type, opts, remaining);
+ remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
+ synack_type, opts, remaining);
return MAX_TCP_OPTION_SPACE - remaining;
}
@@ -1157,6 +1197,7 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
unsigned int eff_sacks;
opts->options = 0;
+ opts->bpf_opt_len = 0;
/* Better than switch (key.type) as it has static branches */
if (tcp_key_is_md5(key)) {
@@ -1244,6 +1285,15 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
size = MAX_TCP_OPTION_SPACE - remaining;
}
+ if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) {
+ unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
+
+ remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+ remaining);
+
+ size = MAX_TCP_OPTION_SPACE - remaining;
+ }
+
return size;
}