From: Kuniyuki Iwashima <kuniyu@google.com>
To: Alexei Starovoitov <ast@kernel.org>,
Daniel Borkmann <daniel@iogearbox.net>,
Andrii Nakryiko <andrii@kernel.org>,
Martin KaFai Lau <martin.lau@linux.dev>,
Eduard Zingerman <eddyz87@gmail.com>,
Kumar Kartikeya Dwivedi <memxor@gmail.com>
Cc: "Amery Hung" <ameryhung@gmail.com>,
"Yonghong Song" <yonghong.song@linux.dev>,
"John Fastabend" <john.fastabend@gmail.com>,
"Stanislav Fomichev" <sdf@fomichev.me>,
"Eric Dumazet" <edumazet@kernel.org>,
"Neal Cardwell" <ncardwell@google.com>,
"Willem de Bruijn" <willemb@google.com>,
"Tenzin Ukyab" <ukyab@berkeley.edu>,
"Clément Léger" <cleger@meta.com>,
"Kuniyuki Iwashima" <kuniyu@google.com>,
"Kuniyuki Iwashima" <kuni1840@gmail.com>,
bpf@vger.kernel.org, netdev@vger.kernel.org
Subject: [PATCH v4 bpf-next 04/10] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
Date: Tue, 6 Oct 2026 19:24:30 +0000 [thread overview]
Message-ID: <20261006192601.1875100-5-kuniyu@google.com> (raw)
In-Reply-To: <20261006192601.1875100-1-kuniyu@google.com>
bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
incoming / outgoing skb.
bpf_tcp_ops.rtt is called once per RTT, which is every
incoming skb in ping-pong workloads like tcp_rr.
Even attaching NULL callbacks in the fast path hurts performance.
Let's guard them (and write_hdr_opt) with the new per-socket flags.
__bpf_tcp_ops_call() and bpf_tcp_ops_call_flag() are added to check
cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) first and avoid accessing
tp->bpf_tcp_ops_flags when no bpf_tcp_ops is attached.
Note that bpf_tcp_ops_hdr_opt_len() has 3 callers and previously
checked the static key both in tcp_established_options() and inside
bpf_tcp_ops_call(). Now the static key is checked once at the
beginning of bpf_tcp_ops_hdr_opt_len(), and the redundant check in
tcp_established_options() is removed.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
v4: Check static key first and then flags
---
include/net/tcp.h | 44 ++++++++++++++++++++++++++++---------------
net/ipv4/tcp_input.c | 12 +++++++++++-
net/ipv4/tcp_output.c | 34 ++++++++++++++++-----------------
3 files changed, 57 insertions(+), 33 deletions(-)
diff --git a/include/net/tcp.h b/include/net/tcp.h
index 55ad9db99596..b7c0f1a8797a 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3064,22 +3064,33 @@ struct bpf_tcp_ops {
u32 opt_off);
};
-#define bpf_tcp_ops_call(op, sk, ...) \
+#define __bpf_tcp_ops_call(op, sk, ...) \
do { \
- if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { \
- const struct bpf_prog_array_item *item; \
- const struct bpf_tcp_ops *tcp_ops; \
- struct cgroup *cgrp; \
+ const struct bpf_prog_array_item *item; \
+ const struct bpf_tcp_ops *tcp_ops; \
+ struct cgroup *cgrp; \
\
- cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data); \
- rcu_read_lock_dont_migrate(); \
- bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp, \
- CGROUP_TCP_SOCK_OPS) { \
- if (tcp_ops->op) \
- tcp_ops->op(sk, ##__VA_ARGS__); \
- } \
- rcu_read_unlock_migrate(); \
+ cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data); \
+ rcu_read_lock_dont_migrate(); \
+ bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp, \
+ CGROUP_TCP_SOCK_OPS) { \
+ if (tcp_ops->op) \
+ tcp_ops->op(sk, ##__VA_ARGS__); \
} \
+ rcu_read_unlock_migrate(); \
+} while (0)
+
+#define bpf_tcp_ops_call(op, sk, ...) \
+do { \
+ if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) \
+ __bpf_tcp_ops_call(op, sk, ##__VA_ARGS__); \
+} while (0)
+
+#define bpf_tcp_ops_call_flag(op, flag, sk, ...) \
+do { \
+ if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) && \
+ BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), flag)) \
+ __bpf_tcp_ops_call(op, sk, ##__VA_ARGS__); \
} while (0)
#define bpf_tcp_ops_call_int(op, init_retval, sk, ...) \
@@ -3115,7 +3126,9 @@ do { \
})
#else
-#define bpf_tcp_ops_call(op, sk, ...) do { } while (0)
+#define __bpf_tcp_ops_call(op, sk, ...) do { } while (0)
+#define bpf_tcp_ops_call(op, sk, ...) do { } while (0)
+#define bpf_tcp_ops_call_flag(op, flag, sk, ...) do { } while (0)
#define bpf_tcp_ops_call_int(op, init_retval, sk, ...) (init_retval)
#endif
@@ -3150,7 +3163,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
{
if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
- bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
+
+ bpf_tcp_ops_call_flag(rtt, RTT, sk, mrtt, srtt);
}
#if IS_ENABLED(CONFIG_SMC)
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 209db8effcc4..4478d3f3d4b0 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -210,6 +210,11 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
{
+ const struct tcp_sock *tp;
+
+ if (!cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS))
+ return;
+
switch (sk->sk_state) {
case TCP_SYN_RECV:
case TCP_SYN_SENT:
@@ -217,7 +222,12 @@ static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
return;
}
- bpf_tcp_ops_call(parse_hdr, sk, skb);
+ tp = tcp_sk(sk);
+
+ if ((tp->rx_opt.saw_unknown &&
+ BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) ||
+ BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL))
+ __bpf_tcp_ops_call(parse_hdr, sk, skb);
}
static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb,
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index 3ac465513bf5..8770f3084efe 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -581,8 +581,8 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
* writer's bytes). The writer finds the append point by scanning from
* first_opt_off + nr_written to the first NOP.
*/
- bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
- first_opt_off + nr_written);
+ bpf_tcp_ops_call_flag(write_hdr_opt, WRITE_HDR_OPT, sk, skb, req,
+ syn_skb, synack_type, first_opt_off + nr_written);
}
#else
static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -613,11 +613,13 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
{
unsigned int remaining_out = remaining, reserved;
- if (!remaining)
- return 0;
+ if (!cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) ||
+ !BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT)||
+ !remaining)
+ return remaining;
/* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
- bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
+ __bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
reserved = remaining - remaining_out;
if (!reserved)
@@ -1193,8 +1195,9 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
struct tcp_key *key)
{
struct tcp_sock *tp = tcp_sk(sk);
- unsigned int size = 0;
unsigned int eff_sacks;
+ unsigned int remaining;
+ unsigned int size = 0;
opts->options = 0;
opts->bpf_opt_len = 0;
@@ -1224,10 +1227,10 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
* left.
*/
if (sk_is_mptcp(sk)) {
- unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
bool has_ts = opts->options & OPTION_TS;
int opt_size;
+ remaining = MAX_TCP_OPTION_SPACE - size;
opts->mptcp.drop_ts = 0;
opt_size = mptcp_established_options(sk, skb, remaining, has_ts,
@@ -1244,7 +1247,8 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
eff_sacks = tp->rx_opt.num_sacks + tp->rx_opt.dsack;
if (unlikely(eff_sacks)) {
- const unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
+ remaining = MAX_TCP_OPTION_SPACE - size;
+
if (likely(remaining >= TCPOLEN_SACK_BASE_ALIGNED +
TCPOLEN_SACK_PERBLOCK)) {
opts->num_sack_blocks =
@@ -1277,22 +1281,18 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
if (unlikely(BPF_SOCK_OPS_TEST_FLAG(tp,
BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG))) {
- unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
-
+ remaining = MAX_TCP_OPTION_SPACE - size;
remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
remaining);
size = MAX_TCP_OPTION_SPACE - remaining;
}
- if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) {
- unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
-
- remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
- remaining);
+ remaining = MAX_TCP_OPTION_SPACE - size;
+ remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+ remaining);
- size = MAX_TCP_OPTION_SPACE - remaining;
- }
+ size = MAX_TCP_OPTION_SPACE - remaining;
return size;
}
--
2.56.0.360.g66cac248cb-goog
next prev parent reply other threads:[~2026-10-06 19:26 UTC|newest]
Thread overview: 34+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-06 19:24 [PATCH v4 bpf-next 00/10] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-06 19:24 ` [PATCH v4 bpf-next 01/10] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-06 23:09 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 02/10] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-06 22:04 ` Stanislav Fomichev
2026-10-06 23:10 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 03/10] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
2026-10-06 22:04 ` Stanislav Fomichev
2026-10-06 23:12 ` Amery Hung
2026-10-06 19:24 ` Kuniyuki Iwashima [this message]
2026-10-06 22:05 ` [PATCH v4 bpf-next 04/10] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Stanislav Fomichev
2026-10-06 23:25 ` Amery Hung
2026-10-07 2:48 ` Kuniyuki Iwashima
2026-10-07 22:36 ` Amery Hung
2026-10-08 2:10 ` Kuniyuki Iwashima
2026-10-06 19:24 ` [PATCH v4 bpf-next 05/10] bpf: tcp: Check BPF_SOCK_OPS_TEST_FLAG() after cgroup_bpf_enabled(CGROUP_SOCK_OPS) Kuniyuki Iwashima
2026-10-06 20:11 ` bot+bpf-ci
2026-10-06 22:05 ` Stanislav Fomichev
2026-10-07 22:38 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 06/10] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-06 22:06 ` Stanislav Fomichev
2026-10-06 23:26 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 07/10] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
2026-10-07 22:43 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 08/10] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
2026-10-06 22:06 ` Stanislav Fomichev
2026-10-07 22:43 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 09/10] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
2026-10-07 23:06 ` Amery Hung
2026-10-07 23:21 ` Kuniyuki Iwashima
2026-10-07 23:42 ` Amery Hung
2026-10-06 19:24 ` [PATCH v4 bpf-next 10/10] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-06 22:06 ` Stanislav Fomichev
2026-10-07 13:11 ` [PATCH v4 bpf-next 00/10] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Jakub Sitnicki
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261006192601.1875100-5-kuniyu@google.com \
--to=kuniyu@google.com \
--cc=ameryhung@gmail.com \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=cleger@meta.com \
--cc=daniel@iogearbox.net \
--cc=eddyz87@gmail.com \
--cc=edumazet@kernel.org \
--cc=john.fastabend@gmail.com \
--cc=kuni1840@gmail.com \
--cc=martin.lau@linux.dev \
--cc=memxor@gmail.com \
--cc=ncardwell@google.com \
--cc=netdev@vger.kernel.org \
--cc=sdf@fomichev.me \
--cc=ukyab@berkeley.edu \
--cc=willemb@google.com \
--cc=yonghong.song@linux.dev \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox