All of lore.kernel.org
 help / color / mirror / Atom feed
From: Kuniyuki Iwashima <kuniyu@google.com>
To: Alexei Starovoitov <ast@kernel.org>,
	Daniel Borkmann <daniel@iogearbox.net>,
	 Andrii Nakryiko <andrii@kernel.org>,
	Martin KaFai Lau <martin.lau@linux.dev>,
	 Eduard Zingerman <eddyz87@gmail.com>,
	Kumar Kartikeya Dwivedi <memxor@gmail.com>
Cc: "Amery Hung" <ameryhung@gmail.com>,
	"Yonghong Song" <yonghong.song@linux.dev>,
	"John Fastabend" <john.fastabend@gmail.com>,
	"Stanislav Fomichev" <sdf@fomichev.me>,
	"Eric Dumazet" <edumazet@kernel.org>,
	"Neal Cardwell" <ncardwell@google.com>,
	"Willem de Bruijn" <willemb@google.com>,
	"Tenzin Ukyab" <ukyab@berkeley.edu>,
	"Clément Léger" <cleger@meta.com>,
	"Kuniyuki Iwashima" <kuniyu@google.com>,
	"Kuniyuki Iwashima" <kuni1840@gmail.com>,
	bpf@vger.kernel.org, netdev@vger.kernel.org,
	"Emil Tsalapatis" <emil@etsalapatis.com>
Subject: [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat.
Date: Mon,  5 Oct 2026 15:40:50 +0000	[thread overview]
Message-ID: <20261005154533.4147685-9-kuniyu@google.com> (raw)
In-Reply-To: <20261005154533.4147685-1-kuniyu@google.com>

bpf_tcp_ops.{enqueue,dequeue}_rcvq() were added to parse skb and
adjust sk->sk_rcvlowat dynamically to suppress unnecessary wakeups.

Let's add a new kfunc to set sk->sk_rcvlowat.

Negative values are clamped to INT_MAX, consistent with SO_RCVLOWAT.

For enqueue_rcvq(), wakeup is set to false because:

  * tcp_data_ready() is always called after the hooks in
    tcp_queue_rcv() and tcp_ofo_queue().

  * when tcp_fastopen_add_skb() is called for TFO SYN, the socket is
    not yet accept()ed, and when called for TFO SYN+ACK, the socket
    is woken up by sk->sk_state_change() anyway.

For dequeue_rcvq(), wakeup is set to true because tcp_data_ready()
is not called in that path.

An alternative would be to support bpf_setsockopt() for these
hooks.

However, that approach involves excessive conditionals and an
unnecessary memcpy(), costs we do not want to pay for every skb
in the TCP fast path.

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
Acked-by: Stanislav Fomichev <sdf@fomichev.me>
Tested-by: Clément Léger <cleger@meta.com>
Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com>
---
 net/ipv4/bpf_tcp_ops.c | 46 ++++++++++++++++++++++++++++++++++++++++++
 1 file changed, 46 insertions(+)

diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index c73d3478a8d5..4e4846121ce7 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -361,12 +361,31 @@ __bpf_kfunc int bpf_tcp_ops_set_flags(struct tcp_sock *tp, u32 enable, u32 disab
 	return 0;
 }
 
+__bpf_kfunc int bpf_tcp_ops_set_rcvlowat(struct sock *sk, int rcvlowat,
+					 const struct bpf_prog_aux *aux)
+{
+	u32 moff = aux->attach_st_ops_member_off;
+	bool wakeup = false;
+
+	if (moff == offsetof(struct bpf_tcp_ops, dequeue_rcvq))
+		wakeup = true;
+
+	if (rcvlowat < 0)
+		rcvlowat = INT_MAX;
+
+	return __tcp_set_rcvlowat(sk, rcvlowat, wakeup);
+}
+
 __bpf_kfunc_end_defs();
 
 BTF_KFUNCS_START(bpf_tcp_ops_set_flags_kfunc_set)
 BTF_ID_FLAGS(func, bpf_tcp_ops_set_flags)
 BTF_KFUNCS_END(bpf_tcp_ops_set_flags_kfunc_set)
 
+BTF_KFUNCS_START(bpf_tcp_ops_set_rcvlowat_kfunc_set)
+BTF_ID_FLAGS(func, bpf_tcp_ops_set_rcvlowat, KF_IMPLICIT_ARGS)
+BTF_KFUNCS_END(bpf_tcp_ops_set_rcvlowat_kfunc_set)
+
 static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog,
 					      u32 kfunc_id)
 {
@@ -386,6 +405,31 @@ static const struct btf_kfunc_id_set bpf_tcp_ops_set_flags_kfunc_id_set = {
 	.filter = bpf_tcp_ops_set_flags_kfunc_filter,
 };
 
+static int bpf_tcp_ops_set_rcvlowat_kfunc_filter(const struct bpf_prog *prog,
+						 u32 kfunc_id)
+{
+	u32 moff;
+
+	if (!btf_id_set8_contains(&bpf_tcp_ops_set_rcvlowat_kfunc_set, kfunc_id))
+		return 0;
+
+	if (prog->aux->st_ops != &bpf_tcp_ops)
+		return -EACCES;
+
+	moff = prog->aux->attach_st_ops_member_off;
+	if (moff != offsetof(struct bpf_tcp_ops, enqueue_rcvq) &&
+	    moff != offsetof(struct bpf_tcp_ops, dequeue_rcvq))
+		return -EACCES;
+
+	return 0;
+}
+
+static const struct btf_kfunc_id_set bpf_tcp_ops_set_rcvlowat_kfunc_id_set = {
+	.owner = THIS_MODULE,
+	.set = &bpf_tcp_ops_set_rcvlowat_kfunc_set,
+	.filter = bpf_tcp_ops_set_rcvlowat_kfunc_filter,
+};
+
 static int __init __bpf_tcp_ops_init(void)
 {
 	int ret;
@@ -394,6 +438,8 @@ static int __init __bpf_tcp_ops_init(void)
 					&bpf_tcp_ops_set_flags_kfunc_id_set);
 	ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT,
 					       &bpf_tcp_ops_set_flags_kfunc_id_set);
+	ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
+					       &bpf_tcp_ops_set_rcvlowat_kfunc_id_set);
 	ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
 
 	return ret;
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-10-05 15:45 UTC|newest]

Thread overview: 22+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-05 15:55   ` sashiko-bot
2026-10-05 16:27   ` bot+bpf-ci
2026-10-05 17:26     ` Kuniyuki Iwashima
2026-10-05 18:30   ` Stanislav Fomichev
2026-10-05 18:37     ` Kuniyuki Iwashima
2026-10-05 22:47       ` Stanislav Fomichev
2026-10-05 23:31         ` Kuniyuki Iwashima
2026-10-05 21:19   ` Amery Hung
2026-10-05 21:23     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Kuniyuki Iwashima
2026-10-05 16:27   ` bot+bpf-ci
2026-10-05 19:10   ` Amery Hung
2026-10-05 19:13     ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
2026-10-05 15:40 ` Kuniyuki Iwashima [this message]
2026-10-05 15:40 ` [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005154533.4147685-9-kuniyu@google.com \
    --to=kuniyu@google.com \
    --cc=ameryhung@gmail.com \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=bpf@vger.kernel.org \
    --cc=cleger@meta.com \
    --cc=daniel@iogearbox.net \
    --cc=eddyz87@gmail.com \
    --cc=edumazet@kernel.org \
    --cc=emil@etsalapatis.com \
    --cc=john.fastabend@gmail.com \
    --cc=kuni1840@gmail.com \
    --cc=martin.lau@linux.dev \
    --cc=memxor@gmail.com \
    --cc=ncardwell@google.com \
    --cc=netdev@vger.kernel.org \
    --cc=sdf@fomichev.me \
    --cc=ukyab@berkeley.edu \
    --cc=willemb@google.com \
    --cc=yonghong.song@linux.dev \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.