Netdev List
 help / color / mirror / Atom feed
From: Kuniyuki Iwashima <kuniyu@google.com>
To: Alexei Starovoitov <ast@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	 Andrii Nakryiko <andrii@kernel.org>,
	Martin KaFai Lau <martin.lau@linux.dev>,
	 Eduard Zingerman <eddyz87@gmail.com>,
	Kumar Kartikeya Dwivedi <memxor@gmail.com>
Cc: "Amery Hung" <ameryhung@gmail.com>,
	"Yonghong Song" <yonghong.song@linux.dev>,
	"John Fastabend" <john.fastabend@gmail.com>,
	"Stanislav Fomichev" <sdf@fomichev.me>,
	"Eric Dumazet" <edumazet@kernel.org>,
	"Neal Cardwell" <ncardwell@google.com>,
	"Willem de Bruijn" <willemb@google.com>,
	"Tenzin Ukyab" <ukyab@berkeley.edu>,
	"Clément Léger" <cleger@meta.com>,
	"Kuniyuki Iwashima" <kuniyu@google.com>,
	"Kuniyuki Iwashima" <kuni1840@gmail.com>,
	bpf@vger.kernel.org, netdev@vger.kernel.org
Subject: [PATCH v5 bpf-next 04/10] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
Date: Thu,  8 Oct 2026 03:15:24 +0000	[thread overview]
Message-ID: <20261008031604.256498-5-kuniyu@google.com> (raw)
In-Reply-To: <20261008031604.256498-1-kuniyu@google.com>

bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
incoming / outgoing skb.

bpf_tcp_ops.rtt is called once per RTT, which is every
incoming skb in ping-pong workloads like tcp_rr.

Even attaching NULL callbacks in the fast path hurts performance.

Let's guard them (and write_hdr_opt) with the new per-socket flags.

__bpf_tcp_ops_call() and bpf_tcp_ops_call_flag() are added to check
cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) first and avoid accessing
tp->bpf_tcp_ops_flags when no bpf_tcp_ops is attached.

Note that bpf_tcp_ops_hdr_opt_len() has 3 callers and previously
checked the static key both in tcp_established_options() and inside
bpf_tcp_ops_call().  Now the static key is checked once at the
beginning of bpf_tcp_ops_hdr_opt_len(), and the redundant check in
tcp_established_options() is removed.

tcp_bpf_hdr_opt_len() must use __always_inline, not just
inline, otherwise Clang inlines bpf_tcp_ops_hdr_opt_len() to
tcp_bpf_hdr_opt_len() instead.

  $ llvm-objdump -S -D --disassemble=tcp_established_options vmlinux
  ...
  ; 	if (unlikely(BPF_SOCK_OPS_TEST_FLAG(tp,
  ffffffff8251270e: 41 f6 87 20 0c 00 00 40      	testb	$0x40, 0xc20(%r15)
  ffffffff82512716: 75 47                	jne	0xffffffff8251275f <tcp_established_options+0x1ff>
  ; 	asm goto(ARCH_STATIC_BRANCH_ASM("%c0 + %c1", "%l[l_yes]")
  ffffffff82512718: 0f 1f 44 00 00       	nopl	(%rax,%rax)
  ; 	return size;
  ffffffff8251271d: 44 89 e0             	movl	%r12d, %eax
  ffffffff82512720: 5b                   	popq	%rbx
  ffffffff82512721: 41 5c                	popq	%r12
  ffffffff82512723: 41 5e                	popq	%r14
  ffffffff82512725: 41 5f                	popq	%r15
  ffffffff82512727: 5d                   	popq	%rbp
  ffffffff82512728: 2e e9 42 03 43 00    	jmp	0xffffffff82942a70 <__x86_return_thunk>

bpf_skops_write_hdr_opt() is still not free, but it can be
optimised later if needed.

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
v5: Always inline static key to tcp_established_options()
v4: Check static key first and then flags
---
 include/net/tcp.h     | 44 ++++++++++++++++++++++------------
 net/ipv4/tcp_input.c  | 12 +++++++++-
 net/ipv4/tcp_output.c | 55 ++++++++++++++++++++++++++-----------------
 3 files changed, 74 insertions(+), 37 deletions(-)

diff --git a/include/net/tcp.h b/include/net/tcp.h
index 55ad9db99596..b7c0f1a8797a 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3064,22 +3064,33 @@ struct bpf_tcp_ops {
 			      u32 opt_off);
 };
 
-#define bpf_tcp_ops_call(op, sk, ...)					\
+#define __bpf_tcp_ops_call(op, sk, ...)					\
 do {									\
-	if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) {			\
-		const struct bpf_prog_array_item *item;			\
-		const struct bpf_tcp_ops *tcp_ops;			\
-		struct cgroup *cgrp;					\
+	const struct bpf_prog_array_item *item;				\
+	const struct bpf_tcp_ops *tcp_ops;				\
+	struct cgroup *cgrp;						\
 									\
-		cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data);		\
-		rcu_read_lock_dont_migrate();				\
-		bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp,	\
-					      CGROUP_TCP_SOCK_OPS) {	\
-			if (tcp_ops->op)				\
-				tcp_ops->op(sk, ##__VA_ARGS__);		\
-		}							\
-		rcu_read_unlock_migrate();				\
+	cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data);			\
+	rcu_read_lock_dont_migrate();					\
+	bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp,		\
+				      CGROUP_TCP_SOCK_OPS) {		\
+		if (tcp_ops->op)					\
+			tcp_ops->op(sk, ##__VA_ARGS__);			\
 	}								\
+	rcu_read_unlock_migrate();					\
+} while (0)
+
+#define bpf_tcp_ops_call(op, sk, ...)					\
+do {									\
+	if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS))			\
+		__bpf_tcp_ops_call(op, sk, ##__VA_ARGS__);		\
+} while (0)
+
+#define bpf_tcp_ops_call_flag(op, flag, sk, ...)			\
+do {									\
+	if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) &&			\
+	    BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), flag))			\
+		__bpf_tcp_ops_call(op, sk, ##__VA_ARGS__);		\
 } while (0)
 
 #define bpf_tcp_ops_call_int(op, init_retval, sk, ...)			\
@@ -3115,7 +3126,9 @@ do {									\
 })
 
 #else
-#define bpf_tcp_ops_call(op, sk, ...)		do { } while (0)
+#define __bpf_tcp_ops_call(op, sk, ...)			do { } while (0)
+#define bpf_tcp_ops_call(op, sk, ...)			do { } while (0)
+#define bpf_tcp_ops_call_flag(op, flag, sk, ...)	do { } while (0)
 #define bpf_tcp_ops_call_int(op, init_retval, sk, ...)	(init_retval)
 #endif
 
@@ -3150,7 +3163,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
 {
 	if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
 		tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
-	bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
+
+	bpf_tcp_ops_call_flag(rtt, RTT, sk, mrtt, srtt);
 }
 
 #if IS_ENABLED(CONFIG_SMC)
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 209db8effcc4..4478d3f3d4b0 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -210,6 +210,11 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
 
 static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
 {
+	const struct tcp_sock *tp;
+
+	if (!cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS))
+		return;
+
 	switch (sk->sk_state) {
 	case TCP_SYN_RECV:
 	case TCP_SYN_SENT:
@@ -217,7 +222,12 @@ static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
 		return;
 	}
 
-	bpf_tcp_ops_call(parse_hdr, sk, skb);
+	tp = tcp_sk(sk);
+
+	if ((tp->rx_opt.saw_unknown &&
+	     BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) ||
+	    BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL))
+		__bpf_tcp_ops_call(parse_hdr, sk, skb);
 }
 
 static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb,
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index 3ac465513bf5..313dfe70a386 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -581,8 +581,8 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
 	 * writer's bytes). The writer finds the append point by scanning from
 	 * first_opt_off + nr_written to the first NOP.
 	 */
-	bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
-			 first_opt_off + nr_written);
+	bpf_tcp_ops_call_flag(write_hdr_opt, WRITE_HDR_OPT, sk, skb, req,
+			      syn_skb, synack_type, first_opt_off + nr_written);
 }
 #else
 static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -613,11 +613,12 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
 {
 	unsigned int remaining_out = remaining, reserved;
 
-	if (!remaining)
-		return 0;
+	if (!BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT) ||
+	    !remaining)
+		return remaining;
 
 	/* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
-	bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
+	__bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
 
 	reserved = remaining - remaining_out;
 	if (!reserved)
@@ -630,6 +631,21 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
 	return remaining - reserved;
 }
 
+static __always_inline u32 tcp_bpf_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
+					       struct request_sock *req,
+					       struct sk_buff *syn_skb,
+					       enum tcp_synack_type synack_type,
+					       struct tcp_out_options *opts,
+					       u32 remaining)
+{
+	if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS))
+		remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, req, syn_skb,
+						    synack_type, opts,
+						    remaining);
+
+	return remaining;
+}
+
 static __be32 *process_tcp_ao_options(struct tcp_sock *tp,
 				      const struct tcp_request_sock *tcprsk,
 				      struct tcp_out_options *opts,
@@ -1089,8 +1105,8 @@ static unsigned int tcp_syn_options(struct sock *sk, struct sk_buff *skb,
 
 	remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
 					  remaining);
-	remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
-					    remaining);
+	remaining = tcp_bpf_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+					remaining);
 
 	return MAX_TCP_OPTION_SPACE - remaining;
 }
@@ -1179,8 +1195,8 @@ static unsigned int tcp_synack_options(const struct sock *sk,
 
 	remaining = bpf_skops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
 					  synack_type, opts, remaining);
-	remaining = bpf_tcp_ops_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
-					    synack_type, opts, remaining);
+	remaining = tcp_bpf_hdr_opt_len((struct sock *)sk, skb, req, syn_skb,
+					synack_type, opts, remaining);
 
 	return MAX_TCP_OPTION_SPACE - remaining;
 }
@@ -1193,8 +1209,9 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
 					struct tcp_key *key)
 {
 	struct tcp_sock *tp = tcp_sk(sk);
-	unsigned int size = 0;
 	unsigned int eff_sacks;
+	unsigned int remaining;
+	unsigned int size = 0;
 
 	opts->options = 0;
 	opts->bpf_opt_len = 0;
@@ -1224,10 +1241,10 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
 	 * left.
 	 */
 	if (sk_is_mptcp(sk)) {
-		unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
 		bool has_ts = opts->options & OPTION_TS;
 		int opt_size;
 
+		remaining = MAX_TCP_OPTION_SPACE - size;
 		opts->mptcp.drop_ts = 0;
 
 		opt_size = mptcp_established_options(sk, skb, remaining, has_ts,
@@ -1244,7 +1261,8 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
 
 	eff_sacks = tp->rx_opt.num_sacks + tp->rx_opt.dsack;
 	if (unlikely(eff_sacks)) {
-		const unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
+		remaining = MAX_TCP_OPTION_SPACE - size;
+
 		if (likely(remaining >= TCPOLEN_SACK_BASE_ALIGNED +
 					TCPOLEN_SACK_PERBLOCK)) {
 			opts->num_sack_blocks =
@@ -1277,22 +1295,17 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb
 
 	if (unlikely(BPF_SOCK_OPS_TEST_FLAG(tp,
 					    BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG))) {
-		unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
-
+		remaining = MAX_TCP_OPTION_SPACE - size;
 		remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
 						  remaining);
 
 		size = MAX_TCP_OPTION_SPACE - remaining;
 	}
 
-	if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) {
-		unsigned int remaining = MAX_TCP_OPTION_SPACE - size;
+	remaining = tcp_bpf_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
+					MAX_TCP_OPTION_SPACE - size);
 
-		remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts,
-						    remaining);
-
-		size = MAX_TCP_OPTION_SPACE - remaining;
-	}
+	size = MAX_TCP_OPTION_SPACE - remaining;
 
 	return size;
 }
-- 
2.56.0.360.g66cac248cb-goog


  parent reply	other threads:[~2026-10-08  3:16 UTC|newest]

Thread overview: 16+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-08  3:15 [PATCH v5 bpf-next 00/10] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 01/10] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 02/10] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 03/10] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
2026-10-08  3:15 ` Kuniyuki Iwashima [this message]
2026-10-08  4:02   ` [PATCH v5 bpf-next 04/10] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag bot+bpf-ci
2026-10-08 22:11   ` Stanislav Fomichev
2026-10-08  3:15 ` [PATCH v5 bpf-next 05/10] bpf: tcp: Check BPF_SOCK_OPS_TEST_FLAG() after cgroup_bpf_enabled(CGROUP_SOCK_OPS) Kuniyuki Iwashima
2026-10-08 16:56   ` Amery Hung
2026-10-08 22:11   ` Stanislav Fomichev
2026-10-08  3:15 ` [PATCH v5 bpf-next 06/10] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 07/10] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 08/10] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 09/10] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
2026-10-08  3:15 ` [PATCH v5 bpf-next 10/10] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-08 17:16   ` Amery Hung

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261008031604.256498-5-kuniyu@google.com \
    --to=kuniyu@google.com \
    --cc=ameryhung@gmail.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=cleger@meta.com \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=edumazet@kernel.org \
    --cc=john.fastabend@gmail.com \
    --cc=kuni1840@gmail.com \
    --cc=martin.lau@linux.dev \
    --cc=memxor@gmail.com \
    --cc=ncardwell@google.com \
    --cc=netdev@vger.kernel.org \
    --cc=sdf@fomichev.me \
    --cc=ukyab@berkeley.edu \
    --cc=willemb@google.com \
    --cc=yonghong.song@linux.dev \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox