BPF List
 help / color / mirror / Atom feed
From: Kuniyuki Iwashima <kuniyu@google.com>
To: Alexei Starovoitov <ast@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	 Andrii Nakryiko <andrii@kernel.org>,
	Martin KaFai Lau <martin.lau@linux.dev>,
	 Eduard Zingerman <eddyz87@gmail.com>,
	Kumar Kartikeya Dwivedi <memxor@gmail.com>
Cc: "Amery Hung" <ameryhung@gmail.com>,
	"Yonghong Song" <yonghong.song@linux.dev>,
	"John Fastabend" <john.fastabend@gmail.com>,
	"Stanislav Fomichev" <sdf@fomichev.me>,
	"Eric Dumazet" <edumazet@kernel.org>,
	"Neal Cardwell" <ncardwell@google.com>,
	"Willem de Bruijn" <willemb@google.com>,
	"Tenzin Ukyab" <ukyab@berkeley.edu>,
	"Clément Léger" <cleger@meta.com>,
	"Kuniyuki Iwashima" <kuniyu@google.com>,
	"Kuniyuki Iwashima" <kuni1840@gmail.com>,
	bpf@vger.kernel.org, netdev@vger.kernel.org
Subject: [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
Date: Mon,  5 Oct 2026 15:40:46 +0000	[thread overview]
Message-ID: <20261005154533.4147685-5-kuniyu@google.com> (raw)
In-Reply-To: <20261005154533.4147685-1-kuniyu@google.com>

bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
incoming / outgoing skb.

bpf_tcp_ops.rtt is called once per RTT, which is every
incoming skb in ping-pong workloads like tcp_rr.

Even attaching NULL callbacks in the fast path hurts performance.

Let's guard them (and write_hdr_opt) with the new per-socket flags.

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
 include/net/tcp.h     |  3 ++-
 net/ipv4/tcp_input.c  |  7 +++++++
 net/ipv4/tcp_output.c | 20 +++++++++++---------
 3 files changed, 20 insertions(+), 10 deletions(-)

diff --git a/include/net/tcp.h b/include/net/tcp.h
index f69091f5e01f..48487135c076 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
 {
 	if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
 		tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
-	bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
+	if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
+		bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
 }
 
 #if IS_ENABLED(CONFIG_SMC)
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 222c7bf80542..79d721215f52 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -210,6 +210,13 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
 
 static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
 {
+	const struct tcp_sock *tp = tcp_sk(sk);
+
+	if (!(tp->rx_opt.saw_unknown &&
+	      BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) &&
+	    !BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL))
+		return;
+
 	switch (sk->sk_state) {
 	case TCP_SYN_RECV:
 	case TCP_SYN_SENT:
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index 908944d409d6..3ec26ad347ef 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -576,13 +576,14 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
 		memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP,
 		       max_opt_len - nr_written);
 
-	/*
-	 * bpf_tcp_ops portion is NOP-filled (everything past the sockops
-	 * writer's bytes). The writer finds the append point by scanning from
-	 * first_opt_off + nr_written to the first NOP.
-	 */
-	bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
-			 first_opt_off + nr_written);
+	if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT)) {
+		/* bpf_tcp_ops portion is NOP-filled (everything past the sockops
+		 * writer's bytes). The writer finds the append point by scanning from
+		 * first_opt_off + nr_written to the first NOP.
+		 */
+		bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
+				 first_opt_off + nr_written);
+	}
 }
 #else
 static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -613,8 +614,9 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
 {
 	unsigned int remaining_out = remaining, reserved;
 
-	if (!remaining)
-		return 0;
+	if (!BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT) ||
+	    !remaining)
+		return remaining;
 
 	/* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
 	bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-10-05 15:45 UTC|newest]

Thread overview: 22+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-05 15:55   ` sashiko-bot
2026-10-05 16:27   ` bot+bpf-ci
2026-10-05 17:26     ` Kuniyuki Iwashima
2026-10-05 18:30   ` Stanislav Fomichev
2026-10-05 18:37     ` Kuniyuki Iwashima
2026-10-05 22:47       ` Stanislav Fomichev
2026-10-05 23:31         ` Kuniyuki Iwashima
2026-10-05 21:19   ` Amery Hung
2026-10-05 21:23     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
2026-10-05 15:40 ` Kuniyuki Iwashima [this message]
2026-10-05 16:27   ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag bot+bpf-ci
2026-10-05 19:10   ` Amery Hung
2026-10-05 19:13     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005154533.4147685-5-kuniyu@google.com \
    --to=kuniyu@google.com \
    --cc=ameryhung@gmail.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=cleger@meta.com \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=edumazet@kernel.org \
    --cc=john.fastabend@gmail.com \
    --cc=kuni1840@gmail.com \
    --cc=martin.lau@linux.dev \
    --cc=memxor@gmail.com \
    --cc=ncardwell@google.com \
    --cc=netdev@vger.kernel.org \
    --cc=sdf@fomichev.me \
    --cc=ukyab@berkeley.edu \
    --cc=willemb@google.com \
    --cc=yonghong.song@linux.dev \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox