All of lore.kernel.org
 help / color / mirror / Atom feed
From: Kuniyuki Iwashima <kuniyu@google.com>
To: Alexei Starovoitov <ast@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	 Andrii Nakryiko <andrii@kernel.org>,
	Martin KaFai Lau <martin.lau@linux.dev>,
	 Eduard Zingerman <eddyz87@gmail.com>,
	Kumar Kartikeya Dwivedi <memxor@gmail.com>
Cc: "Amery Hung" <ameryhung@gmail.com>,
	"Yonghong Song" <yonghong.song@linux.dev>,
	"John Fastabend" <john.fastabend@gmail.com>,
	"Stanislav Fomichev" <sdf@fomichev.me>,
	"Eric Dumazet" <edumazet@kernel.org>,
	"Neal Cardwell" <ncardwell@google.com>,
	"Willem de Bruijn" <willemb@google.com>,
	"Tenzin Ukyab" <ukyab@berkeley.edu>,
	"Clément Léger" <cleger@meta.com>,
	"Kuniyuki Iwashima" <kuniyu@google.com>,
	"Kuniyuki Iwashima" <kuni1840@gmail.com>,
	bpf@vger.kernel.org, netdev@vger.kernel.org
Subject: [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
Date: Mon,  5 Oct 2026 15:40:46 +0000	[thread overview]
Message-ID: <20261005154533.4147685-5-kuniyu@google.com> (raw)
In-Reply-To: <20261005154533.4147685-1-kuniyu@google.com>

bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
incoming / outgoing skb.

bpf_tcp_ops.rtt is called once per RTT, which is every
incoming skb in ping-pong workloads like tcp_rr.

Even attaching NULL callbacks in the fast path hurts performance.

Let's guard them (and write_hdr_opt) with the new per-socket flags.

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
 include/net/tcp.h     |  3 ++-
 net/ipv4/tcp_input.c  |  7 +++++++
 net/ipv4/tcp_output.c | 20 +++++++++++---------
 3 files changed, 20 insertions(+), 10 deletions(-)

diff --git a/include/net/tcp.h b/include/net/tcp.h
index f69091f5e01f..48487135c076 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
 {
 	if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
 		tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
-	bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
+	if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
+		bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
 }
 
 #if IS_ENABLED(CONFIG_SMC)
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 222c7bf80542..79d721215f52 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -210,6 +210,13 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
 
 static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
 {
+	const struct tcp_sock *tp = tcp_sk(sk);
+
+	if (!(tp->rx_opt.saw_unknown &&
+	      BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) &&
+	    !BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL))
+		return;
+
 	switch (sk->sk_state) {
 	case TCP_SYN_RECV:
 	case TCP_SYN_SENT:
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index 908944d409d6..3ec26ad347ef 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -576,13 +576,14 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
 		memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP,
 		       max_opt_len - nr_written);
 
-	/*
-	 * bpf_tcp_ops portion is NOP-filled (everything past the sockops
-	 * writer's bytes). The writer finds the append point by scanning from
-	 * first_opt_off + nr_written to the first NOP.
-	 */
-	bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
-			 first_opt_off + nr_written);
+	if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT)) {
+		/* bpf_tcp_ops portion is NOP-filled (everything past the sockops
+		 * writer's bytes). The writer finds the append point by scanning from
+		 * first_opt_off + nr_written to the first NOP.
+		 */
+		bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
+				 first_opt_off + nr_written);
+	}
 }
 #else
 static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -613,8 +614,9 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
 {
 	unsigned int remaining_out = remaining, reserved;
 
-	if (!remaining)
-		return 0;
+	if (!BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT) ||
+	    !remaining)
+		return remaining;
 
 	/* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
 	bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-10-05 15:45 UTC|newest]

Thread overview: 22+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-05 15:55   ` sashiko-bot
2026-10-05 16:27   ` bot+bpf-ci
2026-10-05 17:26     ` Kuniyuki Iwashima
2026-10-05 18:30   ` Stanislav Fomichev
2026-10-05 18:37     ` Kuniyuki Iwashima
2026-10-05 22:47       ` Stanislav Fomichev
2026-10-05 23:31         ` Kuniyuki Iwashima
2026-10-05 21:19   ` Amery Hung
2026-10-05 21:23     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
2026-10-05 15:40 ` Kuniyuki Iwashima [this message]
2026-10-05 16:27   ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag bot+bpf-ci
2026-10-05 19:10   ` Amery Hung
2026-10-05 19:13     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005154533.4147685-5-kuniyu@google.com \
    --to=kuniyu@google.com \
    --cc=ameryhung@gmail.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=cleger@meta.com \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=edumazet@kernel.org \
    --cc=john.fastabend@gmail.com \
    --cc=kuni1840@gmail.com \
    --cc=martin.lau@linux.dev \
    --cc=memxor@gmail.com \
    --cc=ncardwell@google.com \
    --cc=netdev@vger.kernel.org \
    --cc=sdf@fomichev.me \
    --cc=ukyab@berkeley.edu \
    --cc=willemb@google.com \
    --cc=yonghong.song@linux.dev \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.