summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--include/linux/filter.h5
-rw-r--r--include/net/tcp.h43
-rw-r--r--include/uapi/linux/bpf.h35
-rw-r--r--net/core/filter.c32
-rw-r--r--net/ipv4/bpf_tcp_ops.c144
-rw-r--r--net/ipv4/tcp_input.c13
-rw-r--r--net/ipv4/tcp_output.c96
-rw-r--r--tools/include/uapi/linux/bpf.h35
8 files changed, 341 insertions, 62 deletions
diff --git a/include/linux/filter.h b/include/linux/filter.h
index b17222db2efc..422284b4fa96 100644
--- a/include/linux/filter.h
+++ b/include/linux/filter.h
@@ -1930,6 +1930,11 @@ static __always_inline long __bpf_xdp_redirect_map(struct bpf_map *map, u64 inde
return XDP_REDIRECT;
}
+int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ void *search_res, u32 len, u64 flags);
+int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ const void *from, u32 len, u64 flags);
+
#ifdef CONFIG_NET
int __bpf_skb_load_bytes(const struct sk_buff *skb, u32 offset, void *to, u32 len);
int __bpf_skb_store_bytes(struct sk_buff *skb, u32 offset, const void *from,
diff --git a/include/net/tcp.h b/include/net/tcp.h
index af3747043778..d61ee00052e3 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3011,6 +3011,48 @@ struct bpf_tcp_ops {
/* Called on listen(2), right after the socket enters TCP_LISTEN. */
void (*listen)(struct sock *sk);
+
+ /*
+ * Parse the TCP header options of an incoming skb received on an
+ * established connection. Use bpf_dynptr_from_skb()/bpf_skb_load_bytes()
+ * to access the options.
+ */
+ void (*parse_hdr)(struct sock *sk, struct sk_buff *skb);
+
+ /*
+ * Reserve space in the outgoing TCP header for options to be written
+ * later by write_hdr_opt(). Call bpf_reserve_hdr_opt() to reserve bytes.
+ *
+ * @skb: outgoing packet. NULL when called from tcp_current_mss()
+ * (MSS sizing).
+ * @req: request_sock on the synack path; NULL otherwise.
+ * @syn_skb: incoming SYN on the synack path; NULL otherwise.
+ * @synack_type: TCP_SYNACK_COOKIE indicates a stateless syncookie.
+ * @remaining: pointer to the size of space still available; cast it
+ * using bpf_rdonly_cast() before dereferencing.
+ */
+ void (*hdr_opt_len)(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req, struct sk_buff *syn_skb,
+ enum tcp_synack_type synack_type,
+ unsigned int *remaining);
+
+ /*
+ * Write header options into the space reserved earlier by hdr_opt_len().
+ * Use bpf_store_hdr_opt() to write; it appends within the reserved window
+ * shared with legacy SOCKOPS.
+ *
+ * @skb: outgoing packet.
+ * @req: request_sock on the synack path; NULL otherwise.
+ * @syn_skb: incoming SYN on the synack path; NULL otherwise.
+ * @synack_type: TCP_SYNACK_COOKIE indicates a stateless syncookie.
+ * @opt_off: offset in the outgoing @skb's TCP header where the
+ * bpf_tcp_ops portion of the reserved window begins, i.e. after
+ * the kernel and legacy options.
+ */
+ void (*write_hdr_opt)(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req, struct sk_buff *syn_skb,
+ enum tcp_synack_type synack_type,
+ u32 opt_off);
};
#define bpf_tcp_ops_call(op, sk, ...) \
@@ -3062,6 +3104,7 @@ do { \
} \
__retval; \
})
+
#else
#define bpf_tcp_ops_call(op, sk, ...) do { } while (0)
#define bpf_tcp_ops_call_int(op, init_retval, sk, ...) (init_retval)
diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
index dabe01cd3def..6330b7d745c5 100644
--- a/include/uapi/linux/bpf.h
+++ b/include/uapi/linux/bpf.h
@@ -4867,15 +4867,18 @@ union bpf_attr {
* The non-negative copied *buf* length equal to or less than
* *size* on success, or a negative error in case of failure.
*
- * long bpf_load_hdr_opt(struct bpf_sock_ops *skops, void *searchby_res, u32 len, u64 flags)
+ * long bpf_load_hdr_opt(void *ctx, void *searchby_res, u32 len, u64 flags)
* Description
* Load header option. Support reading a particular TCP header
- * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**).
+ * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). For the
+ * **bpf_tcp_ops** struct_ops, this helper can be called from the
+ * **parse_hdr**\ () and **write_hdr_opt**\ () operators.
*
- * If *flags* is 0, it will search the option from the
- * *skops*\ **->skb_data**. The comment in **struct bpf_sock_ops**
- * has details on what skb_data contains under different
- * *skops*\ **->op**.
+ * If *flags* is 0, it will search the option from the packet
+ * associated with the current operation. For
+ * **BPF_PROG_TYPE_SOCK_OPS**, the comment in
+ * **struct bpf_sock_ops** has details on what skb_data
+ * contains under different *op*.
*
* The first byte of the *searchby_res* specifies the
* kind that it wants to search.
@@ -4908,6 +4911,8 @@ union bpf_attr {
*
* * **BPF_LOAD_HDR_OPT_TCP_SYN** to search from the
* saved_syn packet or the just-received syn packet.
+ * Not supported by the **bpf_tcp_ops** struct_ops, which
+ * rejects all flags.
*
* Return
* > 0 when found, the header option is copied to *searchby_res*.
@@ -4928,9 +4933,9 @@ union bpf_attr {
* packet.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
- * long bpf_store_hdr_opt(struct bpf_sock_ops *skops, const void *from, u32 len, u64 flags)
+ * long bpf_store_hdr_opt(void *ctx, const void *from, u32 len, u64 flags)
* Description
* Store header option. The data will be copied
* from buffer *from* with length *len* to the TCP header.
@@ -4946,7 +4951,9 @@ union bpf_attr {
* by searching the same option in the outgoing skb.
*
* This helper can only be called during
- * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**.
+ * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**, or from the
+ * **write_hdr_opt**\ () operator of the **bpf_tcp_ops**
+ * struct_ops.
*
* Return
* 0 on success, or negative error in case of failure:
@@ -4961,9 +4968,9 @@ union bpf_attr {
* **-EFAULT** on failure to parse the existing header options.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
- * long bpf_reserve_hdr_opt(struct bpf_sock_ops *skops, u32 len, u64 flags)
+ * long bpf_reserve_hdr_opt(void *ctx, u32 len, u64 flags)
* Description
* Reserve *len* bytes for the bpf header option. The
* space will be used by **bpf_store_hdr_opt**\ () later in
@@ -4973,7 +4980,9 @@ union bpf_attr {
* the total number of bytes will be reserved.
*
* This helper can only be called during
- * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**.
+ * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**, or from the
+ * **hdr_opt_len**\ () operator of the **bpf_tcp_ops**
+ * struct_ops.
*
* Return
* 0 on success, or negative error in case of failure:
@@ -4983,7 +4992,7 @@ union bpf_attr {
* **-ENOSPC** if there is not enough space in the header.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
* void *bpf_inode_storage_get(struct bpf_map *map, void *inode, void *value, u64 flags)
* Description
diff --git a/net/core/filter.c b/net/core/filter.c
index 47b7a8f73069..5feb99884682 100644
--- a/net/core/filter.c
+++ b/net/core/filter.c
@@ -8053,17 +8053,14 @@ static const u8 *bpf_search_tcp_opt(const u8 *op, const u8 *opend,
return ERR_PTR(-ENOMSG);
}
-BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
- void *, search_res, u32, len, u64, flags)
+int __bpf_sock_ops_load_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ void *search_res, u32 len, u64 flags)
{
bool eol, load_syn = flags & BPF_LOAD_HDR_OPT_TCP_SYN;
const u8 *op, *opend, *magic, *search = search_res;
u8 search_kind, search_len, copy_len, magic_len;
int ret;
- if (!is_locked_tcp_sock_ops(bpf_sock))
- return -EOPNOTSUPP;
-
/* 2 byte is the minimal option len except TCPOPT_NOP and
* TCPOPT_EOL which are useless for the bpf prog to learn
* and this helper disallow loading them also.
@@ -8124,6 +8121,15 @@ BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
return ret;
}
+BPF_CALL_4(bpf_sock_ops_load_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
+ void *, search_res, u32, len, u64, flags)
+{
+ if (!is_locked_tcp_sock_ops(bpf_sock))
+ return -EOPNOTSUPP;
+
+ return __bpf_sock_ops_load_hdr_opt(bpf_sock, search_res, len, flags);
+}
+
static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = {
.func = bpf_sock_ops_load_hdr_opt,
.gpl_only = false,
@@ -8134,17 +8140,14 @@ static const struct bpf_func_proto bpf_sock_ops_load_hdr_opt_proto = {
.arg4_type = ARG_ANYTHING,
};
-BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
- const void *, from, u32, len, u64, flags)
+int __bpf_sock_ops_store_hdr_opt(struct bpf_sock_ops_kern *bpf_sock,
+ const void *from, u32 len, u64 flags)
{
u8 new_kind, new_kind_len, magic_len = 0, *opend;
const u8 *op, *new_op, *magic = NULL;
struct sk_buff *skb;
bool eol;
- if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB)
- return -EPERM;
-
if (len < 2 || flags)
return -EINVAL;
@@ -8202,6 +8205,15 @@ BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
return 0;
}
+BPF_CALL_4(bpf_sock_ops_store_hdr_opt, struct bpf_sock_ops_kern *, bpf_sock,
+ const void *, from, u32, len, u64, flags)
+{
+ if (bpf_sock->op != BPF_SOCK_OPS_WRITE_HDR_OPT_CB)
+ return -EPERM;
+
+ return __bpf_sock_ops_store_hdr_opt(bpf_sock, from, len, flags);
+}
+
static const struct bpf_func_proto bpf_sock_ops_store_hdr_opt_proto = {
.func = bpf_sock_ops_store_hdr_opt,
.gpl_only = false,
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index 3febbc8dd1a0..681fed642999 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -4,6 +4,7 @@
#include <linux/bpf.h>
#include <linux/btf_ids.h>
#include <linux/bpf_verifier.h>
+#include <linux/filter.h>
#include <net/bpf_sk_storage.h>
#include <net/tcp.h>
@@ -55,6 +56,26 @@ static void listen_stub(struct sock *sk)
{
}
+static void parse_hdr_stub(struct sock *sk, struct sk_buff *skb)
+{
+}
+
+static void hdr_opt_len_stub(struct sock *sk, struct sk_buff *skb__nullable,
+ struct request_sock *req__nullable,
+ struct sk_buff *syn_skb__nullable,
+ enum tcp_synack_type synack_type,
+ unsigned int *remaining)
+{
+}
+
+static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req__nullable,
+ struct sk_buff *syn_skb__nullable,
+ enum tcp_synack_type synack_type,
+ u32 opt_off)
+{
+}
+
static struct bpf_tcp_ops __bpf_tcp_ops = {
.timeout_init = timeout_init_stub,
.rwnd_init = rwnd_init_stub,
@@ -66,6 +87,104 @@ static struct bpf_tcp_ops __bpf_tcp_ops = {
.retrans = retrans_stub,
.connect = connect_stub,
.listen = listen_stub,
+ .parse_hdr = parse_hdr_stub,
+ .hdr_opt_len = hdr_opt_len_stub,
+ .write_hdr_opt = write_hdr_opt_stub,
+};
+
+BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from,
+ u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ struct sk_buff *skb = (void *)(unsigned long)args[1];
+ struct bpf_sock_ops_kern sock_ops = {};
+ u32 opt_off = args[5];
+ u8 *op, *opend;
+
+ /*
+ * bpf_tcp_ops does not keep track of the end of the written TCP header
+ * options, so search for it every time the helper is called. The free
+ * space is NOP-filled, so a TCPOPT_NOP ends the search rather than being
+ * skipped as in a normal option walk in sockops.
+ */
+ op = skb->data + opt_off;
+ opend = skb->data + tcp_hdrlen(skb);
+ while (op < opend && *op != TCPOPT_NOP) {
+ if (*op == TCPOPT_EOL || op + 1 >= opend || op[1] < 2)
+ break;
+ op += op[1];
+ }
+
+ sock_ops.skb = skb;
+ sock_ops.skb_data_end = op;
+ sock_ops.remaining_opt_len = opend - op;
+
+ return __bpf_sock_ops_store_hdr_opt(&sock_ops, from, len, flags);
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_store_hdr_opt_proto = {
+ .func = bpf_tcp_ops_store_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_PTR_TO_MEM | MEM_RDONLY,
+ .arg3_type = ARG_MEM_SIZE,
+ .arg4_type = ARG_ANYTHING,
+};
+
+BPF_CALL_4(bpf_tcp_ops_load_hdr_opt, void *, ctx, void *, search_res,
+ u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ struct sk_buff *skb = (void *)(unsigned long)args[1];
+ struct bpf_sock_ops_kern sock_ops = {};
+
+ /*
+ * No flags supported. In particular BPF_LOAD_HDR_OPT_TCP_SYN, which
+ * loads from the saved SYN, is not available because bpf_tcp_ops has no
+ * carrier to track the SYN source across the hooks.
+ */
+ if (flags)
+ return -EINVAL;
+
+ sock_ops.skb = skb;
+ sock_ops.skb_data_end = skb->data + tcp_hdrlen(skb);
+
+ return __bpf_sock_ops_load_hdr_opt(&sock_ops, search_res, len, flags);
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_load_hdr_opt_proto = {
+ .func = bpf_tcp_ops_load_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_PTR_TO_MEM | MEM_WRITE,
+ .arg3_type = ARG_MEM_SIZE,
+ .arg4_type = ARG_ANYTHING,
+};
+
+BPF_CALL_3(bpf_tcp_ops_reserve_hdr_opt, void *, ctx, u32, len, u64, flags)
+{
+ u64 *args = ctx;
+ unsigned int *remaining = (void *)(unsigned long)args[5];
+
+ if (flags || len < 2)
+ return -EINVAL;
+
+ if (len > *remaining)
+ return -ENOSPC;
+
+ *remaining -= len;
+ return 0;
+}
+
+static const struct bpf_func_proto bpf_tcp_ops_reserve_hdr_opt_proto = {
+ .func = bpf_tcp_ops_reserve_hdr_opt,
+ .gpl_only = false,
+ .ret_type = RET_INTEGER,
+ .arg1_type = ARG_PTR_TO_CTX,
+ .arg2_type = ARG_ANYTHING,
+ .arg3_type = ARG_ANYTHING,
};
BPF_CALL_0(bpf_tcp_ops_get_retval)
@@ -102,14 +221,20 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
case BPF_FUNC_sk_storage_delete:
return &bpf_sk_storage_delete_proto;
case BPF_FUNC_setsockopt:
- /* The listener is not locked. */
+ /* The sk may be an unlocked listener (synack path) or NULL
+ * fullsock; disable for members that can run unlocked.
+ */
if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init))
+ moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
+ moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
return NULL;
return &bpf_sk_setsockopt_proto;
case BPF_FUNC_getsockopt:
if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init))
+ moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
+ moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
return NULL;
return &bpf_sk_getsockopt_proto;
case BPF_FUNC_get_retval:
@@ -117,6 +242,19 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
moff == offsetof(struct bpf_tcp_ops, rwnd_init))
return &bpf_tcp_ops_get_retval_proto;
return NULL;
+ case BPF_FUNC_reserve_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, hdr_opt_len))
+ return &bpf_tcp_ops_reserve_hdr_opt_proto;
+ return NULL;
+ case BPF_FUNC_load_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, parse_hdr) ||
+ moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
+ return &bpf_tcp_ops_load_hdr_opt_proto;
+ return NULL;
+ case BPF_FUNC_store_hdr_opt:
+ if (moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
+ return &bpf_tcp_ops_store_hdr_opt_proto;
+ return NULL;
default:
return bpf_base_func_proto(func_id, prog);
}
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 69e6f3925073..6ac6f9d5b6c3 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -208,6 +208,18 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
}
#endif
+static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
+{
+ switch (sk->sk_state) {
+ case TCP_SYN_RECV:
+ case TCP_SYN_SENT:
+ case TCP_LISTEN:
+ return;
+ }
+
+ bpf_tcp_ops_call(parse_hdr, sk, skb);
+}
+
static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb,
unsigned int len)
{
@@ -6461,6 +6473,7 @@ syn_challenge:
pass:
bpf_skops_parse_hdr(sk, skb);
+ bpf_tcp_ops_parse_hdr(sk, skb);
return true;
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index f75a5a01d621..908944d409d6 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -536,43 +536,53 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
enum tcp_synack_type synack_type,
struct tcp_out_options *opts)
{
- u8 first_opt_off, nr_written, max_opt_len = opts->bpf_opt_len;
- struct bpf_sock_ops_kern sock_ops;
- int err;
+ u8 first_opt_off, nr_written = 0, max_opt_len = opts->bpf_opt_len;
if (likely(!max_opt_len))
return;
- memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp));
+ first_opt_off = tcp_hdrlen(skb) - max_opt_len;
- sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB;
+ if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk),
+ BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG)) {
+ struct bpf_sock_ops_kern sock_ops;
+ int err;
- if (req) {
- sock_ops.sk = (struct sock *)req;
- sock_ops.syn_skb = syn_skb;
- } else {
- sock_owned_by_me(sk);
+ memset(&sock_ops, 0, offsetof(struct bpf_sock_ops_kern, temp));
- sock_ops.is_fullsock = 1;
- sock_ops.is_locked_tcp_sock = 1;
- sock_ops.sk = sk;
- }
+ sock_ops.op = BPF_SOCK_OPS_WRITE_HDR_OPT_CB;
- sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type);
- sock_ops.remaining_opt_len = max_opt_len;
- first_opt_off = tcp_hdrlen(skb) - max_opt_len;
- bpf_skops_init_skb(&sock_ops, skb, first_opt_off);
+ if (req) {
+ sock_ops.sk = (struct sock *)req;
+ sock_ops.syn_skb = syn_skb;
+ } else {
+ sock_owned_by_me(sk);
- err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk);
+ sock_ops.is_fullsock = 1;
+ sock_ops.is_locked_tcp_sock = 1;
+ sock_ops.sk = sk;
+ }
- if (err)
- nr_written = 0;
- else
- nr_written = max_opt_len - sock_ops.remaining_opt_len;
+ sock_ops.args[0] = bpf_skops_write_hdr_opt_arg0(skb, synack_type);
+ sock_ops.remaining_opt_len = max_opt_len;
+ bpf_skops_init_skb(&sock_ops, skb, first_opt_off);
+
+ err = BPF_CGROUP_RUN_PROG_SOCK_OPS_SK(&sock_ops, sk);
+ if (!err)
+ nr_written = max_opt_len - sock_ops.remaining_opt_len;
+ }
if (nr_written < max_opt_len)
memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP,
max_opt_len - nr_written);
+
+ /*
+ * bpf_tcp_ops portion is NOP-filled (everything past the sockops
+ * writer's bytes). The writer finds the append point by scanning from
+ * first_opt_off + nr_written to the first NOP.
+ */
+ bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
+ first_opt_off + nr_written);
}
#else
static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -594,6 +604,32 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
}
#endif
+static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
+ struct request_sock *req,
+ struct sk_buff *syn_skb,
+ enum tcp_synack_type synack_type,
+ struct tcp_out_options *opts,
+ u32 remaining)
+{
+ unsigned int remaining_out = remaining, reserved;
+
+ if (!remaining)
+ return 0;
+
+ /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
+ bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
+
+ reserved = remaining - remaining_out;
+ if (!reserved)
+ return remaining;
+
+ /* round up to 4 bytes */
+ reserved = (reserved + 3) & ~3;
+
+ opts->bpf_opt_len += reserved;
+ return remaining - reserved;
+}
+
static __be32 *process_tcp_ao_options(struct tcp_sock *tp,
const struct tcp_request_sock *tcprsk,
struct tcp_out_options *opts,
@@ -1053,6 +1089,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb,
remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
remaining);
+ remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+ remaining);
return MAX_TCP_OPTION_SPACE - remaining;
}
@@ -1141,6 +1179,8 @@ static unsigned int tcp_synack_options(const struct sock *sk,
remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
synack_type, opts, remaining);
+ remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
+ synack_type, opts, remaining);
return MAX_TCP_OPTION_SPACE - remaining;
}
@@ -1157,6 +1197,7 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
unsigned int eff_sacks;
opts->options = 0;
+ opts->bpf_opt_len = 0;
/* Better than switch (key.type) as it has static branches */
if (tcp_key_is_md5(key)) {
@@ -1244,6 +1285,15 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
size = MAX_TCP_OPTION_SPACE - remaining;
}
+ if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) {
+ unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
+
+ remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+ remaining);
+
+ size = MAX_TCP_OPTION_SPACE - remaining;
+ }
+
return size;
}
diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h
index dabe01cd3def..6330b7d745c5 100644
--- a/tools/include/uapi/linux/bpf.h
+++ b/tools/include/uapi/linux/bpf.h
@@ -4867,15 +4867,18 @@ union bpf_attr {
* The non-negative copied *buf* length equal to or less than
* *size* on success, or a negative error in case of failure.
*
- * long bpf_load_hdr_opt(struct bpf_sock_ops *skops, void *searchby_res, u32 len, u64 flags)
+ * long bpf_load_hdr_opt(void *ctx, void *searchby_res, u32 len, u64 flags)
* Description
* Load header option. Support reading a particular TCP header
- * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**).
+ * option for bpf program (**BPF_PROG_TYPE_SOCK_OPS**). For the
+ * **bpf_tcp_ops** struct_ops, this helper can be called from the
+ * **parse_hdr**\ () and **write_hdr_opt**\ () operators.
*
- * If *flags* is 0, it will search the option from the
- * *skops*\ **->skb_data**. The comment in **struct bpf_sock_ops**
- * has details on what skb_data contains under different
- * *skops*\ **->op**.
+ * If *flags* is 0, it will search the option from the packet
+ * associated with the current operation. For
+ * **BPF_PROG_TYPE_SOCK_OPS**, the comment in
+ * **struct bpf_sock_ops** has details on what skb_data
+ * contains under different *op*.
*
* The first byte of the *searchby_res* specifies the
* kind that it wants to search.
@@ -4908,6 +4911,8 @@ union bpf_attr {
*
* * **BPF_LOAD_HDR_OPT_TCP_SYN** to search from the
* saved_syn packet or the just-received syn packet.
+ * Not supported by the **bpf_tcp_ops** struct_ops, which
+ * rejects all flags.
*
* Return
* > 0 when found, the header option is copied to *searchby_res*.
@@ -4928,9 +4933,9 @@ union bpf_attr {
* packet.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
- * long bpf_store_hdr_opt(struct bpf_sock_ops *skops, const void *from, u32 len, u64 flags)
+ * long bpf_store_hdr_opt(void *ctx, const void *from, u32 len, u64 flags)
* Description
* Store header option. The data will be copied
* from buffer *from* with length *len* to the TCP header.
@@ -4946,7 +4951,9 @@ union bpf_attr {
* by searching the same option in the outgoing skb.
*
* This helper can only be called during
- * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**.
+ * **BPF_SOCK_OPS_WRITE_HDR_OPT_CB**, or from the
+ * **write_hdr_opt**\ () operator of the **bpf_tcp_ops**
+ * struct_ops.
*
* Return
* 0 on success, or negative error in case of failure:
@@ -4961,9 +4968,9 @@ union bpf_attr {
* **-EFAULT** on failure to parse the existing header options.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
- * long bpf_reserve_hdr_opt(struct bpf_sock_ops *skops, u32 len, u64 flags)
+ * long bpf_reserve_hdr_opt(void *ctx, u32 len, u64 flags)
* Description
* Reserve *len* bytes for the bpf header option. The
* space will be used by **bpf_store_hdr_opt**\ () later in
@@ -4973,7 +4980,9 @@ union bpf_attr {
* the total number of bytes will be reserved.
*
* This helper can only be called during
- * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**.
+ * **BPF_SOCK_OPS_HDR_OPT_LEN_CB**, or from the
+ * **hdr_opt_len**\ () operator of the **bpf_tcp_ops**
+ * struct_ops.
*
* Return
* 0 on success, or negative error in case of failure:
@@ -4983,7 +4992,7 @@ union bpf_attr {
* **-ENOSPC** if there is not enough space in the header.
*
* **-EPERM** if the helper cannot be used under the current
- * *skops*\ **->op**.
+ * operation.
*
* void *bpf_inode_storage_get(struct bpf_map *map, void *inode, void *value, u64 flags)
* Description