* [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
` (7 subsequent siblings)
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev, Emil Tsalapatis
Currently, four bpf_tcp_ops callbacks are not allowed to call
bpf_setsockopt() and bpf_getsockopt().
However, the deny-list is fragile, and when a new callback is
added, we might re-open a can of worms. [0][1]
Let's convert it to allow-list.
Note that it can be written as offsetof(,X) <= moff <= offsetof(,Y)
but clang can optimise to similar code anyway.
Link: https://lore.kernel.org/r/20220929070407.965581-5-martin.lau@linux.dev/ #[0]
Link: https://lore.kernel.org/bpf/20260421155804.135786-1-kafai.wan@linux.dev/ #[1]
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com>
Acked-by: Stanislav Fomichev <sdf@fomichev.me>
---
net/ipv4/bpf_tcp_ops.c | 39 ++++++++++++++++++++++++---------------
1 file changed, 24 insertions(+), 15 deletions(-)
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index 681fed642999..1ada3b781bf1 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -210,6 +210,24 @@ const struct bpf_func_proto bpf_tcp_ops_get_retval_proto = {
.ret_type = RET_INTEGER,
};
+static bool is_sockopt_supported(u32 moff)
+{
+ switch (moff) {
+ case offsetof(struct bpf_tcp_ops, active_established):
+ case offsetof(struct bpf_tcp_ops, passive_established):
+ case offsetof(struct bpf_tcp_ops, rto):
+ case offsetof(struct bpf_tcp_ops, rtt):
+ case offsetof(struct bpf_tcp_ops, set_state):
+ case offsetof(struct bpf_tcp_ops, retrans):
+ case offsetof(struct bpf_tcp_ops, connect):
+ case offsetof(struct bpf_tcp_ops, listen):
+ case offsetof(struct bpf_tcp_ops, parse_hdr):
+ return true;
+ }
+
+ return false;
+}
+
static const struct bpf_func_proto *
get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
{
@@ -221,22 +239,13 @@ get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
case BPF_FUNC_sk_storage_delete:
return &bpf_sk_storage_delete_proto;
case BPF_FUNC_setsockopt:
- /* The sk may be an unlocked listener (synack path) or NULL
- * fullsock; disable for members that can run unlocked.
- */
- if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
- moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
- moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
- return NULL;
- return &bpf_sk_setsockopt_proto;
+ if (is_sockopt_supported(moff))
+ return &bpf_sk_setsockopt_proto;
+ return NULL;
case BPF_FUNC_getsockopt:
- if (moff == offsetof(struct bpf_tcp_ops, rwnd_init) ||
- moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
- moff == offsetof(struct bpf_tcp_ops, hdr_opt_len) ||
- moff == offsetof(struct bpf_tcp_ops, write_hdr_opt))
- return NULL;
- return &bpf_sk_getsockopt_proto;
+ if (is_sockopt_supported(moff))
+ return &bpf_sk_getsockopt_proto;
+ return NULL;
case BPF_FUNC_get_retval:
if (moff == offsetof(struct bpf_tcp_ops, timeout_init) ||
moff == offsetof(struct bpf_tcp_ops, rwnd_init))
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 16:27 ` bot+bpf-ci
` (2 more replies)
2026-10-05 15:40 ` [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
` (6 subsequent siblings)
8 siblings, 3 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
The legacy SOCK_OPS guards some hooks with a per-socket flag,
tp->bpf_sock_ops_cb_flags.
In contrast, bpf_tcp_ops was initially designed without per-socket
flags so that users can simply define only the callbacks they need.
However, it turned out that even attaching a bpf_tcp_ops with a NULL
callback incurs measurable overhead in the fast path. [0]
We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
1 bit left for a new opt-in callback, while not all of the 7 existing
opt-in hooks are in the fast path and really need a flag guard.
Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
update it, both of which require sock_owned_by_me(sk):
* bpf_sock_ops_cb_flags_set()
* bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
Both helpers only overwrite the field, which leads to reading and
modifying the flags in the BPF prog and then writing them back
via the helper. This prevents future use from tc or cgroup_skb
hooks where bh_lock_sock() alone cannot prevent races with process
context.
Let's add a new u32 field, tp->bpf_tcp_ops_flags, at the end of the
tcp_sock_read_txrx cacheline group, along with a new kfunc,
bpf_tcp_ops_set_flags().
bpf_tcp_ops_set_flags() takes bitmasks of flags to enable and
disable and updates tp->bpf_tcp_ops_flags atomically via
try_cmpxchg() without relying on lock_sock().
The kfunc is exposed to bpf_tcp_ops and BPF_PROG_TYPE_CGROUP_SOCKOPT.
The first argument is struct tcp_sock * so that bpf_tcp_sock() is
required for BPF_PROG_TYPE_CGROUP_SOCKOPT, but not for bpf_tcp_ops
where struct sock * is promoted to struct tcp_sock * automatically.
The fast-path callbacks will be guarded by the new flags in a later
patch after updating the existing selftest to avoid breaking bisection.
Link: https://lore.kernel.org/netdev/CAMB2axMwBuz3X4Uwn5uzZUqm91EbiMnbQX832wUFOQ44fOwDnQ@mail.gmail.com/ #[0]
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
include/linux/tcp.h | 7 +++++
include/net/tcp.h | 1 +
include/uapi/linux/bpf.h | 8 +++++
net/ipv4/bpf_tcp_ops.c | 56 +++++++++++++++++++++++++++++++++-
net/ipv4/tcp.c | 3 ++
tools/include/uapi/linux/bpf.h | 8 +++++
6 files changed, 82 insertions(+), 1 deletion(-)
diff --git a/include/linux/tcp.h b/include/linux/tcp.h
index 6a8c77719322..24bb751cc65a 100644
--- a/include/linux/tcp.h
+++ b/include/linux/tcp.h
@@ -233,6 +233,13 @@ struct tcp_sock {
is_sack_reneg:1, /* in recovery from loss with SACK reneg? */
is_cwnd_limited:1,/* forward progress limited by snd_cwnd? */
recvmsg_inq : 1;/* Indicate # of bytes in queue upon recvmsg */
+#ifdef CONFIG_BPF
+ u32 bpf_tcp_ops_flags;
+#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) \
+ (READ_ONCE((TP)->bpf_tcp_ops_flags) & BPF_TCP_OPS_FLAG_ ## ARG)
+#else
+#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) (0)
+#endif
__cacheline_group_end(tcp_sock_read_txrx);
/* RX read-mostly hotpath cache lines */
diff --git a/include/net/tcp.h b/include/net/tcp.h
index d61ee00052e3..f69091f5e01f 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -2934,6 +2934,7 @@ static inline int tcp_call_bpf_3arg(struct sock *sk, int op, u32 arg1, u32 arg2,
static inline void tcp_clear_sock_ops_cb_flags(struct sock *sk)
{
tcp_sk(sk)->bpf_sock_ops_cb_flags = 0;
+ WRITE_ONCE(tcp_sk(sk)->bpf_tcp_ops_flags, 0);
}
#else
diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
index 4687c3310996..6ff90b73dd38 100644
--- a/include/uapi/linux/bpf.h
+++ b/include/uapi/linux/bpf.h
@@ -7334,6 +7334,14 @@ enum {
*/
};
+enum {
+ BPF_TCP_OPS_FLAG_RTT = (1 << 0),
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
+ BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
+ BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
+};
+
/* List of TCP states. There is a build check in net/ipv4/tcp.c to detect
* changes between the TCP and BPF versions. Ideally this should never happen.
* If it does, we need to add code to convert them before calling
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index 1ada3b781bf1..5693857e764e 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -328,8 +328,62 @@ static struct bpf_struct_ops bpf_tcp_ops = {
.owner = THIS_MODULE,
};
+__bpf_kfunc_start_defs();
+
+__bpf_kfunc int bpf_tcp_ops_set_flags(struct tcp_sock *tp, u32 enable, u32 disable)
+{
+ u32 old, new;
+
+ if ((enable & disable) || (enable | disable) & ~BPF_TCP_OPS_FLAG_ALL)
+ return -EINVAL;
+
+ old = READ_ONCE(tp->bpf_tcp_ops_flags);
+
+ do {
+ new = (old | enable) & ~disable;
+ if (new == old)
+ break;
+ } while (!try_cmpxchg(&tp->bpf_tcp_ops_flags, &old, new));
+
+ return 0;
+}
+
+__bpf_kfunc_end_defs();
+
+BTF_KFUNCS_START(bpf_tcp_ops_set_flags_kfunc_set)
+BTF_ID_FLAGS(func, bpf_tcp_ops_set_flags)
+BTF_KFUNCS_END(bpf_tcp_ops_set_flags_kfunc_set)
+
+static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog,
+ u32 kfunc_id)
+{
+ if (!btf_id_set8_contains(&bpf_tcp_ops_set_flags_kfunc_set, kfunc_id))
+ return 0;
+
+ if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
+ prog->aux->st_ops != &bpf_tcp_ops)
+ return -EACCES;
+
+ return 0;
+}
+
+static const struct btf_kfunc_id_set bpf_tcp_ops_set_flags_kfunc_id_set = {
+ .owner = THIS_MODULE,
+ .set = &bpf_tcp_ops_set_flags_kfunc_set,
+ .filter = bpf_tcp_ops_set_flags_kfunc_filter,
+};
+
static int __init __bpf_tcp_ops_init(void)
{
- return register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
+ int ret;
+
+ ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
+ &bpf_tcp_ops_set_flags_kfunc_id_set);
+ ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT,
+ &bpf_tcp_ops_set_flags_kfunc_id_set);
+ ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
+
+ return ret;
}
+
late_initcall(__bpf_tcp_ops_init);
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
index 650a2e89950a..fa69961c47d3 100644
--- a/net/ipv4/tcp.c
+++ b/net/ipv4/tcp.c
@@ -5217,6 +5217,9 @@ static void __init tcp_struct_check(void)
CACHELINE_ASSERT_GROUP_MEMBER(struct tcp_sock, tcp_sock_read_txrx, lost_out);
CACHELINE_ASSERT_GROUP_MEMBER(struct tcp_sock, tcp_sock_read_txrx, sacked_out);
CACHELINE_ASSERT_GROUP_MEMBER(struct tcp_sock, tcp_sock_read_txrx, scaling_ratio);
+#ifdef CONFIG_BPF
+ CACHELINE_ASSERT_GROUP_MEMBER(struct tcp_sock, tcp_sock_read_txrx, bpf_tcp_ops_flags);
+#endif
/* RX read-mostly hotpath cache lines */
CACHELINE_ASSERT_GROUP_MEMBER(struct tcp_sock, tcp_sock_read_rx, copied_seq);
diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h
index 4687c3310996..6ff90b73dd38 100644
--- a/tools/include/uapi/linux/bpf.h
+++ b/tools/include/uapi/linux/bpf.h
@@ -7334,6 +7334,14 @@ enum {
*/
};
+enum {
+ BPF_TCP_OPS_FLAG_RTT = (1 << 0),
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
+ BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
+ BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
+};
+
/* List of TCP states. There is a build check in net/ipv4/tcp.c to detect
* changes between the TCP and BPF versions. Ideally this should never happen.
* If it does, we need to add code to convert them before calling
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
@ 2026-10-05 16:27 ` bot+bpf-ci
2026-10-05 17:26 ` Kuniyuki Iwashima
2026-10-05 18:30 ` Stanislav Fomichev
2026-10-05 21:19 ` Amery Hung
2 siblings, 1 reply; 21+ messages in thread
From: bot+bpf-ci @ 2026-10-05 16:27 UTC (permalink / raw)
To: kuniyu, ast, daniel, andrii, martin.lau, eddyz87, memxor
Cc: ameryhung, yonghong.song, john.fastabend, sdf, edumazet,
ncardwell, willemb, ukyab, cleger, kuniyu, kuni1840, bpf, netdev,
ast, andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 3480 bytes --]
> bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
TL;DR: The bpf_tcp_ops_set_flags() kfunc filter seems to let cgroup_skb and
sock_ops programs call it too, not just bpf_tcp_ops and cgroup_sockopt; this
is an API scope issue, not memory safety.
> diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
> --- a/net/ipv4/bpf_tcp_ops.c
> +++ b/net/ipv4/bpf_tcp_ops.c
[ ... ]
> +static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog,
> + u32 kfunc_id)
> +{
> + if (!btf_id_set8_contains(&bpf_tcp_ops_set_flags_kfunc_set, kfunc_id))
> + return 0;
> +
> + if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
> + prog->aux->st_ops != &bpf_tcp_ops)
> + return -EACCES;
> +
> + return 0;
> +}
[ ... ]
> static int __init __bpf_tcp_ops_init(void)
> {
> - return register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
> + int ret;
> +
> + ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
> + &bpf_tcp_ops_set_flags_kfunc_id_set);
> + ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT,
> + &bpf_tcp_ops_set_flags_kfunc_id_set);
The commit message says the kfunc is exposed to bpf_tcp_ops and
BPF_PROG_TYPE_CGROUP_SOCKOPT. Does this registration limit it to those two
program types?
register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT, ...) puts the set
into the shared BTF_KFUNC_HOOK_CGROUP hook. bpf_prog_type_to_kfunc_hook()
in kernel/bpf/btf.c maps CGROUP_SKB, CGROUP_SOCK, CGROUP_DEVICE,
CGROUP_SOCK_ADDR, CGROUP_SOCKOPT, CGROUP_SYSCTL and SOCK_OPS all to that
one hook.
btf_kfunc_is_allowed() looks the kfunc up in that hook and then only runs
the hook filters, and bpf_tcp_ops_set_flags_kfunc_filter() returns 0 for
every program type except a STRUCT_OPS program that is not bpf_tcp_ops:
if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
prog->aux->st_ops != &bpf_tcp_ops)
return -EACCES;
return 0;
That seems to let BPF_PROG_TYPE_CGROUP_SKB and BPF_PROG_TYPE_SOCK_OPS
programs call bpf_tcp_ops_set_flags() as well. Both program types provide
bpf_tcp_sock() (cg_skb_func_proto and sock_ops_func_proto in
net/core/filter.c), and the verifier accepts the returned PTR_TO_TCP_SOCK
as a trusted struct tcp_sock * through reg2btf_ids[].
For cgroup_skb the likely caller is an ingress program running in softirq
context. The commit message mentions cgroup_skb only as a possible future
use.
The cmpxchg update keeps this memory safe, so this is about API scope:
once programs start using the kfunc from these other program types,
narrowing it later would break them.
bpf_qdisc_kfunc_filter() and scx_kfunc_context_filter() both reject the
program types they do not intend to serve. If only bpf_tcp_ops and
CGROUP_SOCKOPT are intended, should the filter reject prog->type values
other than BPF_PROG_TYPE_STRUCT_OPS and BPF_PROG_TYPE_CGROUP_SOCKOPT?
Otherwise, could the commit message list the actual set of program types
that can call the kfunc?
This filter is unchanged through 3cb5919af2b2 ("selftest: bpf: Add test for
bpf_tcp_ops.{enqueue,dequeue}_rcvq()."), so no later patch in the series
changes it.
> + ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
> +
> + return ret;
> }
[ ... ]
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/37336989320
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 16:27 ` bot+bpf-ci
@ 2026-10-05 17:26 ` Kuniyuki Iwashima
0 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 17:26 UTC (permalink / raw)
To: bot+bpf-ci
Cc: ast, daniel, andrii, martin.lau, eddyz87, memxor, ameryhung,
yonghong.song, john.fastabend, sdf, edumazet, ncardwell, willemb,
ukyab, cleger, kuni1840, bpf, netdev, martin.lau, mason,
ihor.solodrai
On Mon, Oct 5, 2026 at 9:27 AM <bot+bpf-ci@kernel.org> wrote:
>
> > bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
>
> TL;DR: The bpf_tcp_ops_set_flags() kfunc filter seems to let cgroup_skb and
> sock_ops programs call it too, not just bpf_tcp_ops and cgroup_sockopt; this
> is an API scope issue, not memory safety.
>
> > diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
> > --- a/net/ipv4/bpf_tcp_ops.c
> > +++ b/net/ipv4/bpf_tcp_ops.c
>
> [ ... ]
>
> > +static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog,
> > + u32 kfunc_id)
> > +{
> > + if (!btf_id_set8_contains(&bpf_tcp_ops_set_flags_kfunc_set, kfunc_id))
> > + return 0;
> > +
> > + if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
> > + prog->aux->st_ops != &bpf_tcp_ops)
> > + return -EACCES;
> > +
> > + return 0;
> > +}
>
> [ ... ]
>
> > static int __init __bpf_tcp_ops_init(void)
> > {
> > - return register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
> > + int ret;
> > +
> > + ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
> > + &bpf_tcp_ops_set_flags_kfunc_id_set);
> > + ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT,
> > + &bpf_tcp_ops_set_flags_kfunc_id_set);
>
> The commit message says the kfunc is exposed to bpf_tcp_ops and
> BPF_PROG_TYPE_CGROUP_SOCKOPT. Does this registration limit it to those two
> program types?
>
> register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT, ...) puts the set
> into the shared BTF_KFUNC_HOOK_CGROUP hook. bpf_prog_type_to_kfunc_hook()
> in kernel/bpf/btf.c maps CGROUP_SKB, CGROUP_SOCK, CGROUP_DEVICE,
> CGROUP_SOCK_ADDR, CGROUP_SOCKOPT, CGROUP_SYSCTL and SOCK_OPS all to that
> one hook.
Oh didn't know this. Although it's still safe with try_cmpxchg(),
I will update the filter like this, maybe with CGROUP_SKB allowed too.
if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
prog->aux->st_ops == &bpf_tcp_ops)
return 0;
if (prog->type == BPF_PROG_TYPE_CGROUP_SOCKOPT)
return 0;
return -EACCES;
>
> btf_kfunc_is_allowed() looks the kfunc up in that hook and then only runs
> the hook filters, and bpf_tcp_ops_set_flags_kfunc_filter() returns 0 for
> every program type except a STRUCT_OPS program that is not bpf_tcp_ops:
>
> if (prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
> prog->aux->st_ops != &bpf_tcp_ops)
> return -EACCES;
>
> return 0;
>
> That seems to let BPF_PROG_TYPE_CGROUP_SKB and BPF_PROG_TYPE_SOCK_OPS
> programs call bpf_tcp_ops_set_flags() as well. Both program types provide
> bpf_tcp_sock() (cg_skb_func_proto and sock_ops_func_proto in
> net/core/filter.c), and the verifier accepts the returned PTR_TO_TCP_SOCK
> as a trusted struct tcp_sock * through reg2btf_ids[].
>
> For cgroup_skb the likely caller is an ingress program running in softirq
> context. The commit message mentions cgroup_skb only as a possible future
> use.
>
> The cmpxchg update keeps this memory safe, so this is about API scope:
> once programs start using the kfunc from these other program types,
> narrowing it later would break them.
>
> bpf_qdisc_kfunc_filter() and scx_kfunc_context_filter() both reject the
> program types they do not intend to serve. If only bpf_tcp_ops and
> CGROUP_SOCKOPT are intended, should the filter reject prog->type values
> other than BPF_PROG_TYPE_STRUCT_OPS and BPF_PROG_TYPE_CGROUP_SOCKOPT?
>
> Otherwise, could the commit message list the actual set of program types
> that can call the kfunc?
>
> This filter is unchanged through 3cb5919af2b2 ("selftest: bpf: Add test for
> bpf_tcp_ops.{enqueue,dequeue}_rcvq()."), so no later patch in the series
> changes it.
>
> > + ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
> > +
> > + return ret;
> > }
>
> [ ... ]
>
>
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
>
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/37336989320
^ permalink raw reply [flat|nested] 21+ messages in thread
* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-05 16:27 ` bot+bpf-ci
@ 2026-10-05 18:30 ` Stanislav Fomichev
2026-10-05 18:37 ` Kuniyuki Iwashima
2026-10-05 21:19 ` Amery Hung
2 siblings, 1 reply; 21+ messages in thread
From: Stanislav Fomichev @ 2026-10-05 18:30 UTC (permalink / raw)
To: Kuniyuki Iwashima
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On 10/05, Kuniyuki Iwashima wrote:
> The legacy SOCK_OPS guards some hooks with a per-socket flag,
> tp->bpf_sock_ops_cb_flags.
>
> In contrast, bpf_tcp_ops was initially designed without per-socket
> flags so that users can simply define only the callbacks they need.
>
> However, it turned out that even attaching a bpf_tcp_ops with a NULL
> callback incurs measurable overhead in the fast path. [0]
>
> We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> 1 bit left for a new opt-in callback, while not all of the 7 existing
> opt-in hooks are in the fast path and really need a flag guard.
>
> Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> update it, both of which require sock_owned_by_me(sk):
>
> * bpf_sock_ops_cb_flags_set()
> * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
Can we return -EINVAL from these when used with sock_ops?
If we keep one flag we can presumably put both tcp_call_bpf_Xarg and
bpf_tcp_ops_call under the same if conditional?
^ permalink raw reply [flat|nested] 21+ messages in thread
* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 18:30 ` Stanislav Fomichev
@ 2026-10-05 18:37 ` Kuniyuki Iwashima
2026-10-05 22:47 ` Stanislav Fomichev
0 siblings, 1 reply; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 18:37 UTC (permalink / raw)
To: Stanislav Fomichev
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 11:30 AM Stanislav Fomichev <sdf.kernel@gmail.com> wrote:
>
> On 10/05, Kuniyuki Iwashima wrote:
> > The legacy SOCK_OPS guards some hooks with a per-socket flag,
> > tp->bpf_sock_ops_cb_flags.
> >
> > In contrast, bpf_tcp_ops was initially designed without per-socket
> > flags so that users can simply define only the callbacks they need.
> >
> > However, it turned out that even attaching a bpf_tcp_ops with a NULL
> > callback incurs measurable overhead in the fast path. [0]
> >
> > We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> > 1 bit left for a new opt-in callback, while not all of the 7 existing
> > opt-in hooks are in the fast path and really need a flag guard.
> >
> > Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> > update it, both of which require sock_owned_by_me(sk):
> >
> > * bpf_sock_ops_cb_flags_set()
> > * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
>
> Can we return -EINVAL from these when used with sock_ops?
bpf_sock_ops_cb_flags_set() is not allowed for bpf_tcp_ops,
so we could add a check in bpf_setsockopt() for bpf_tcp_ops
to disallow TCP_BPF_SOCK_OPS_CB_FLAGS.
But not all of bpf_tcp_ops hooks are opt-in, so we cannot prevent
attaching bpf_tcp_ops to SOCK_OPS-enabled sockets.
> If we keep one flag we can presumably put both tcp_call_bpf_Xarg and
> bpf_tcp_ops_call under the same if conditional?
I assuemd we will use the new flag only for bpf_tcp_ops and will no
longer add a new bit to the sock_ops flag and fainlly we can deprecate
it under a Kconfig like CONFIG_BPF_LEGACY_SOCK_OPS.
^ permalink raw reply [flat|nested] 21+ messages in thread
* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 18:37 ` Kuniyuki Iwashima
@ 2026-10-05 22:47 ` Stanislav Fomichev
2026-10-05 23:31 ` Kuniyuki Iwashima
0 siblings, 1 reply; 21+ messages in thread
From: Stanislav Fomichev @ 2026-10-05 22:47 UTC (permalink / raw)
To: Kuniyuki Iwashima
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On 10/05, Kuniyuki Iwashima wrote:
> On Mon, Oct 5, 2026 at 11:30 AM Stanislav Fomichev <sdf.kernel@gmail.com> wrote:
> >
> > On 10/05, Kuniyuki Iwashima wrote:
> > > The legacy SOCK_OPS guards some hooks with a per-socket flag,
> > > tp->bpf_sock_ops_cb_flags.
> > >
> > > In contrast, bpf_tcp_ops was initially designed without per-socket
> > > flags so that users can simply define only the callbacks they need.
> > >
> > > However, it turned out that even attaching a bpf_tcp_ops with a NULL
> > > callback incurs measurable overhead in the fast path. [0]
> > >
> > > We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> > > 1 bit left for a new opt-in callback, while not all of the 7 existing
> > > opt-in hooks are in the fast path and really need a flag guard.
> > >
> > > Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> > > update it, both of which require sock_owned_by_me(sk):
> > >
> > > * bpf_sock_ops_cb_flags_set()
> > > * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
> >
> > Can we return -EINVAL from these when used with sock_ops?
>
> bpf_sock_ops_cb_flags_set() is not allowed for bpf_tcp_ops,
> so we could add a check in bpf_setsockopt() for bpf_tcp_ops
> to disallow TCP_BPF_SOCK_OPS_CB_FLAGS.
>
> But not all of bpf_tcp_ops hooks are opt-in, so we cannot prevent
> attaching bpf_tcp_ops to SOCK_OPS-enabled sockets.
>
>
> > If we keep one flag we can presumably put both tcp_call_bpf_Xarg and
> > bpf_tcp_ops_call under the same if conditional?
[..]
> I assuemd we will use the new flag only for bpf_tcp_ops and will no
> longer add a new bit to the sock_ops flag and fainlly we can deprecate
> it under a Kconfig like CONFIG_BPF_LEGACY_SOCK_OPS.
Don't think CONFIG_LEGACY is happening :-D I might be overthinking it..
Realistically, maybe something like bubbling up cgroup_bpf_enabled(sock_ops)
is an option as well? So these (old) runtime checks are happening only when
we have sock_ops attached (and/or separate mechanism for struct_ops?).
But if you don't see anything wrong in your benchmark, let's not overcomplicate.
^ permalink raw reply [flat|nested] 21+ messages in thread
* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 22:47 ` Stanislav Fomichev
@ 2026-10-05 23:31 ` Kuniyuki Iwashima
0 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 23:31 UTC (permalink / raw)
To: Stanislav Fomichev
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 3:47 PM Stanislav Fomichev <sdf.kernel@gmail.com> wrote:
>
> On 10/05, Kuniyuki Iwashima wrote:
> > On Mon, Oct 5, 2026 at 11:30 AM Stanislav Fomichev <sdf.kernel@gmail.com> wrote:
> > >
> > > On 10/05, Kuniyuki Iwashima wrote:
> > > > The legacy SOCK_OPS guards some hooks with a per-socket flag,
> > > > tp->bpf_sock_ops_cb_flags.
> > > >
> > > > In contrast, bpf_tcp_ops was initially designed without per-socket
> > > > flags so that users can simply define only the callbacks they need.
> > > >
> > > > However, it turned out that even attaching a bpf_tcp_ops with a NULL
> > > > callback incurs measurable overhead in the fast path. [0]
> > > >
> > > > We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> > > > 1 bit left for a new opt-in callback, while not all of the 7 existing
> > > > opt-in hooks are in the fast path and really need a flag guard.
> > > >
> > > > Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> > > > update it, both of which require sock_owned_by_me(sk):
> > > >
> > > > * bpf_sock_ops_cb_flags_set()
> > > > * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
> > >
> > > Can we return -EINVAL from these when used with sock_ops?
> >
> > bpf_sock_ops_cb_flags_set() is not allowed for bpf_tcp_ops,
> > so we could add a check in bpf_setsockopt() for bpf_tcp_ops
> > to disallow TCP_BPF_SOCK_OPS_CB_FLAGS.
> >
> > But not all of bpf_tcp_ops hooks are opt-in, so we cannot prevent
> > attaching bpf_tcp_ops to SOCK_OPS-enabled sockets.
> >
> >
> > > If we keep one flag we can presumably put both tcp_call_bpf_Xarg and
> > > bpf_tcp_ops_call under the same if conditional?
>
> [..]
>
> > I assuemd we will use the new flag only for bpf_tcp_ops and will no
> > longer add a new bit to the sock_ops flag and fainlly we can deprecate
> > it under a Kconfig like CONFIG_BPF_LEGACY_SOCK_OPS.
>
> Don't think CONFIG_LEGACY is happening :-D I might be overthinking it..
Ah, sorry, I realized I misread the code and thought bpf_tcp_ops shared the
same static key with the legacy one, but they actually use separate keys.
> Realistically, maybe something like bubbling up cgroup_bpf_enabled(sock_ops)
> is an option as well?
So yes, this should be enough,
> So these (old) runtime checks are happening only when
> we have sock_ops attached (and/or separate mechanism for struct_ops?).
and this is already done by the separate static keys, except that some
sock_ops call sites check the flags before the static key.
> But if you don't see anything wrong in your benchmark, let's not overcomplicate.
I thought SOCK_OPS also contributed to the regression by enabling it via
the same static key, but the delta was purely from bpf_tcp_ops. So
I think the current form + Amery's suggestion (moving the flag check
after the static key) would be fine. While at it, I can apply the same change
to SOCK_OPS too.
^ permalink raw reply [flat|nested] 21+ messages in thread
* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
2026-10-05 16:27 ` bot+bpf-ci
2026-10-05 18:30 ` Stanislav Fomichev
@ 2026-10-05 21:19 ` Amery Hung
2026-10-05 21:23 ` Kuniyuki Iwashima
2 siblings, 1 reply; 21+ messages in thread
From: Amery Hung @ 2026-10-05 21:19 UTC (permalink / raw)
To: Kuniyuki Iwashima
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Yonghong Song, John Fastabend, Stanislav Fomichev, Eric Dumazet,
Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 8:45 AM Kuniyuki Iwashima <kuniyu@google.com> wrote:
>
> The legacy SOCK_OPS guards some hooks with a per-socket flag,
> tp->bpf_sock_ops_cb_flags.
>
> In contrast, bpf_tcp_ops was initially designed without per-socket
> flags so that users can simply define only the callbacks they need.
>
> However, it turned out that even attaching a bpf_tcp_ops with a NULL
> callback incurs measurable overhead in the fast path. [0]
>
> We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> 1 bit left for a new opt-in callback, while not all of the 7 existing
> opt-in hooks are in the fast path and really need a flag guard.
>
> Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> update it, both of which require sock_owned_by_me(sk):
>
> * bpf_sock_ops_cb_flags_set()
> * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
>
> Both helpers only overwrite the field, which leads to reading and
> modifying the flags in the BPF prog and then writing them back
> via the helper. This prevents future use from tc or cgroup_skb
> hooks where bh_lock_sock() alone cannot prevent races with process
> context.
>
> Let's add a new u32 field, tp->bpf_tcp_ops_flags, at the end of the
> tcp_sock_read_txrx cacheline group, along with a new kfunc,
> bpf_tcp_ops_set_flags().
>
> bpf_tcp_ops_set_flags() takes bitmasks of flags to enable and
> disable and updates tp->bpf_tcp_ops_flags atomically via
> try_cmpxchg() without relying on lock_sock().
>
> The kfunc is exposed to bpf_tcp_ops and BPF_PROG_TYPE_CGROUP_SOCKOPT.
> The first argument is struct tcp_sock * so that bpf_tcp_sock() is
> required for BPF_PROG_TYPE_CGROUP_SOCKOPT, but not for bpf_tcp_ops
> where struct sock * is promoted to struct tcp_sock * automatically.
>
> The fast-path callbacks will be guarded by the new flags in a later
> patch after updating the existing selftest to avoid breaking bisection.
>
> Link: https://lore.kernel.org/netdev/CAMB2axMwBuz3X4Uwn5uzZUqm91EbiMnbQX832wUFOQ44fOwDnQ@mail.gmail.com/ #[0]
> Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
> ---
> include/linux/tcp.h | 7 +++++
> include/net/tcp.h | 1 +
> include/uapi/linux/bpf.h | 8 +++++
> net/ipv4/bpf_tcp_ops.c | 56 +++++++++++++++++++++++++++++++++-
> net/ipv4/tcp.c | 3 ++
> tools/include/uapi/linux/bpf.h | 8 +++++
> 6 files changed, 82 insertions(+), 1 deletion(-)
>
> diff --git a/include/linux/tcp.h b/include/linux/tcp.h
> index 6a8c77719322..24bb751cc65a 100644
> --- a/include/linux/tcp.h
> +++ b/include/linux/tcp.h
> @@ -233,6 +233,13 @@ struct tcp_sock {
> is_sack_reneg:1, /* in recovery from loss with SACK reneg? */
> is_cwnd_limited:1,/* forward progress limited by snd_cwnd? */
> recvmsg_inq : 1;/* Indicate # of bytes in queue upon recvmsg */
> +#ifdef CONFIG_BPF
> + u32 bpf_tcp_ops_flags;
> +#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) \
> + (READ_ONCE((TP)->bpf_tcp_ops_flags) & BPF_TCP_OPS_FLAG_ ## ARG)
> +#else
> +#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) (0)
> +#endif
> __cacheline_group_end(tcp_sock_read_txrx);
>
> /* RX read-mostly hotpath cache lines */
> diff --git a/include/net/tcp.h b/include/net/tcp.h
> index d61ee00052e3..f69091f5e01f 100644
> --- a/include/net/tcp.h
> +++ b/include/net/tcp.h
> @@ -2934,6 +2934,7 @@ static inline int tcp_call_bpf_3arg(struct sock *sk, int op, u32 arg1, u32 arg2,
> static inline void tcp_clear_sock_ops_cb_flags(struct sock *sk)
> {
> tcp_sk(sk)->bpf_sock_ops_cb_flags = 0;
> + WRITE_ONCE(tcp_sk(sk)->bpf_tcp_ops_flags, 0);
> }
>
> #else
> diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
> index 4687c3310996..6ff90b73dd38 100644
> --- a/include/uapi/linux/bpf.h
> +++ b/include/uapi/linux/bpf.h
> @@ -7334,6 +7334,14 @@ enum {
> */
> };
>
> +enum {
> + BPF_TCP_OPS_FLAG_RTT = (1 << 0),
> + BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
> + BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
> + BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
> + BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
> +};
> +
This series changes several existing bpf_tcp_ops callbacks from
enabled-by-presence to disabled by default. Since this needs a respin,
could you also document the opt-in semantics and update the comments
for the affected members of bpf_tcp_ops to name their gating flags?
It would also be useful to document that the mask is per-socket and
shared by all effective bpf_tcp_ops programs.
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops.
2026-10-05 21:19 ` Amery Hung
@ 2026-10-05 21:23 ` Kuniyuki Iwashima
0 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 21:23 UTC (permalink / raw)
To: Amery Hung
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Yonghong Song, John Fastabend, Stanislav Fomichev, Eric Dumazet,
Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 2:19 PM Amery Hung <ameryhung@gmail.com> wrote:
>
> On Mon, Oct 5, 2026 at 8:45 AM Kuniyuki Iwashima <kuniyu@google.com> wrote:
> >
> > The legacy SOCK_OPS guards some hooks with a per-socket flag,
> > tp->bpf_sock_ops_cb_flags.
> >
> > In contrast, bpf_tcp_ops was initially designed without per-socket
> > flags so that users can simply define only the callbacks they need.
> >
> > However, it turned out that even attaching a bpf_tcp_ops with a NULL
> > callback incurs measurable overhead in the fast path. [0]
> >
> > We could reuse tp->bpf_sock_ops_cb_flags, but it is u8 and has only
> > 1 bit left for a new opt-in callback, while not all of the 7 existing
> > opt-in hooks are in the fast path and really need a flag guard.
> >
> > Moreover, tp->bpf_sock_ops_cb_flags has two bpf helper functions to
> > update it, both of which require sock_owned_by_me(sk):
> >
> > * bpf_sock_ops_cb_flags_set()
> > * bpf_setsockopt(TCP_BPF_SOCK_OPS_CB_FLAGS)
> >
> > Both helpers only overwrite the field, which leads to reading and
> > modifying the flags in the BPF prog and then writing them back
> > via the helper. This prevents future use from tc or cgroup_skb
> > hooks where bh_lock_sock() alone cannot prevent races with process
> > context.
> >
> > Let's add a new u32 field, tp->bpf_tcp_ops_flags, at the end of the
> > tcp_sock_read_txrx cacheline group, along with a new kfunc,
> > bpf_tcp_ops_set_flags().
> >
> > bpf_tcp_ops_set_flags() takes bitmasks of flags to enable and
> > disable and updates tp->bpf_tcp_ops_flags atomically via
> > try_cmpxchg() without relying on lock_sock().
> >
> > The kfunc is exposed to bpf_tcp_ops and BPF_PROG_TYPE_CGROUP_SOCKOPT.
> > The first argument is struct tcp_sock * so that bpf_tcp_sock() is
> > required for BPF_PROG_TYPE_CGROUP_SOCKOPT, but not for bpf_tcp_ops
> > where struct sock * is promoted to struct tcp_sock * automatically.
> >
> > The fast-path callbacks will be guarded by the new flags in a later
> > patch after updating the existing selftest to avoid breaking bisection.
> >
> > Link: https://lore.kernel.org/netdev/CAMB2axMwBuz3X4Uwn5uzZUqm91EbiMnbQX832wUFOQ44fOwDnQ@mail.gmail.com/ #[0]
> > Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
> > ---
> > include/linux/tcp.h | 7 +++++
> > include/net/tcp.h | 1 +
> > include/uapi/linux/bpf.h | 8 +++++
> > net/ipv4/bpf_tcp_ops.c | 56 +++++++++++++++++++++++++++++++++-
> > net/ipv4/tcp.c | 3 ++
> > tools/include/uapi/linux/bpf.h | 8 +++++
> > 6 files changed, 82 insertions(+), 1 deletion(-)
> >
> > diff --git a/include/linux/tcp.h b/include/linux/tcp.h
> > index 6a8c77719322..24bb751cc65a 100644
> > --- a/include/linux/tcp.h
> > +++ b/include/linux/tcp.h
> > @@ -233,6 +233,13 @@ struct tcp_sock {
> > is_sack_reneg:1, /* in recovery from loss with SACK reneg? */
> > is_cwnd_limited:1,/* forward progress limited by snd_cwnd? */
> > recvmsg_inq : 1;/* Indicate # of bytes in queue upon recvmsg */
> > +#ifdef CONFIG_BPF
> > + u32 bpf_tcp_ops_flags;
> > +#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) \
> > + (READ_ONCE((TP)->bpf_tcp_ops_flags) & BPF_TCP_OPS_FLAG_ ## ARG)
> > +#else
> > +#define BPF_TCP_OPS_TEST_FLAG(TP, ARG) (0)
> > +#endif
> > __cacheline_group_end(tcp_sock_read_txrx);
> >
> > /* RX read-mostly hotpath cache lines */
> > diff --git a/include/net/tcp.h b/include/net/tcp.h
> > index d61ee00052e3..f69091f5e01f 100644
> > --- a/include/net/tcp.h
> > +++ b/include/net/tcp.h
> > @@ -2934,6 +2934,7 @@ static inline int tcp_call_bpf_3arg(struct sock *sk, int op, u32 arg1, u32 arg2,
> > static inline void tcp_clear_sock_ops_cb_flags(struct sock *sk)
> > {
> > tcp_sk(sk)->bpf_sock_ops_cb_flags = 0;
> > + WRITE_ONCE(tcp_sk(sk)->bpf_tcp_ops_flags, 0);
> > }
> >
> > #else
> > diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
> > index 4687c3310996..6ff90b73dd38 100644
> > --- a/include/uapi/linux/bpf.h
> > +++ b/include/uapi/linux/bpf.h
> > @@ -7334,6 +7334,14 @@ enum {
> > */
> > };
> >
> > +enum {
> > + BPF_TCP_OPS_FLAG_RTT = (1 << 0),
> > + BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
> > + BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
> > + BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
> > + BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
> > +};
> > +
>
> This series changes several existing bpf_tcp_ops callbacks from
> enabled-by-presence to disabled by default. Since this needs a respin,
> could you also document the opt-in semantics and update the comments
> for the affected members of bpf_tcp_ops to name their gating flags?
>
> It would also be useful to document that the mask is per-socket and
> shared by all effective bpf_tcp_ops programs.
Sure, will add doc for the enum and bpf_tcp_ops members.
^ permalink raw reply [flat|nested] 21+ messages in thread
* [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 1/9] bpf: tcp: Convert deny-list for bpf_{get,set}sockopt() to allow-list Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 2/9] bpf: tcp: Add a new per-socket flag and kfunc for bpf_tcp_ops Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Kuniyuki Iwashima
` (5 subsequent siblings)
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
The next patch guards bpf_tcp_ops callbacks in the fast
path with per-socket flags.
Let's use bpf_tcp_ops_set_flags() to enable flags for
parse_hdr, write_hdr_opt, and hdr_opt_len.
Note that BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL and
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN are used on the
server and client, respectively, for better coverage.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
.../selftests/bpf/progs/bpf_tcp_ops_hdr.c | 20 +++++++++++++++++++
1 file changed, 20 insertions(+)
diff --git a/tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c b/tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c
index f3e3dff13784..706e94d794e4 100644
--- a/tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c
+++ b/tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c
@@ -18,6 +18,24 @@ int found_cnt;
__u8 found_d0;
__u8 found_d1;
+SEC("struct_ops")
+void BPF_PROG(test_listen, struct sock *sk)
+{
+ bpf_tcp_ops_set_flags((struct tcp_sock *)sk,
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL |
+ BPF_TCP_OPS_FLAG_WRITE_HDR_OPT,
+ 0);
+}
+
+SEC("struct_ops")
+void BPF_PROG(test_connect, struct sock *sk)
+{
+ bpf_tcp_ops_set_flags((struct tcp_sock *)sk,
+ BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN |
+ BPF_TCP_OPS_FLAG_WRITE_HDR_OPT,
+ 0);
+}
+
SEC("struct_ops")
void BPF_PROG(test_hdr_opt_len, struct sock *sk, struct sk_buff *skb,
struct request_sock *req, struct sk_buff *syn_skb,
@@ -78,6 +96,8 @@ void BPF_PROG(test_parse_hdr, struct sock *sk, struct sk_buff *skb)
SEC(".struct_ops.link")
struct bpf_tcp_ops test_hdr_ops = {
+ .listen = (void *)test_listen,
+ .connect = (void *)test_connect,
.hdr_opt_len = (void *)test_hdr_opt_len,
.write_hdr_opt = (void *)test_write_hdr_opt,
.parse_hdr = (void *)test_parse_hdr,
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (2 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 3/9] selftest: bpf: Use bpf_tcp_ops_set_flags() in bpf_tcp_ops_hdr.c Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 16:27 ` bot+bpf-ci
2026-10-05 19:10 ` Amery Hung
2026-10-05 15:40 ` [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
` (4 subsequent siblings)
8 siblings, 2 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
incoming / outgoing skb.
bpf_tcp_ops.rtt is called once per RTT, which is every
incoming skb in ping-pong workloads like tcp_rr.
Even attaching NULL callbacks in the fast path hurts performance.
Let's guard them (and write_hdr_opt) with the new per-socket flags.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
include/net/tcp.h | 3 ++-
net/ipv4/tcp_input.c | 7 +++++++
net/ipv4/tcp_output.c | 20 +++++++++++---------
3 files changed, 20 insertions(+), 10 deletions(-)
diff --git a/include/net/tcp.h b/include/net/tcp.h
index f69091f5e01f..48487135c076 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
{
if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
- bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
+ if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
+ bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
}
#if IS_ENABLED(CONFIG_SMC)
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 222c7bf80542..79d721215f52 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -210,6 +210,13 @@ static void bpf_skops_established(struct sock *sk, int bpf_op,
static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb)
{
+ const struct tcp_sock *tp = tcp_sk(sk);
+
+ if (!(tp->rx_opt.saw_unknown &&
+ BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) &&
+ !BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL))
+ return;
+
switch (sk->sk_state) {
case TCP_SYN_RECV:
case TCP_SYN_SENT:
diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
index 908944d409d6..3ec26ad347ef 100644
--- a/net/ipv4/tcp_output.c
+++ b/net/ipv4/tcp_output.c
@@ -576,13 +576,14 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb,
memset(skb->data + first_opt_off + nr_written, TCPOPT_NOP,
max_opt_len - nr_written);
- /*
- * bpf_tcp_ops portion is NOP-filled (everything past the sockops
- * writer's bytes). The writer finds the append point by scanning from
- * first_opt_off + nr_written to the first NOP.
- */
- bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
- first_opt_off + nr_written);
+ if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT)) {
+ /* bpf_tcp_ops portion is NOP-filled (everything past the sockops
+ * writer's bytes). The writer finds the append point by scanning from
+ * first_opt_off + nr_written to the first NOP.
+ */
+ bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type,
+ first_opt_off + nr_written);
+ }
}
#else
static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
@@ -613,8 +614,9 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb,
{
unsigned int remaining_out = remaining, reserved;
- if (!remaining)
- return 0;
+ if (!BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT) ||
+ !remaining)
+ return remaining;
/* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */
bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out);
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
2026-10-05 15:40 ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Kuniyuki Iwashima
@ 2026-10-05 16:27 ` bot+bpf-ci
2026-10-05 19:10 ` Amery Hung
1 sibling, 0 replies; 21+ messages in thread
From: bot+bpf-ci @ 2026-10-05 16:27 UTC (permalink / raw)
To: kuniyu, ast, daniel, andrii, martin.lau, eddyz87, memxor
Cc: ameryhung, yonghong.song, john.fastabend, sdf, edumazet,
ncardwell, willemb, ukyab, cleger, kuniyu, kuni1840, bpf, netdev,
ast, andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 2673 bytes --]
> bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
TL;DR: Documentation only, not a bug: the bpf_tcp_ops kernel-doc for rtt,
parse_hdr, hdr_opt_len and write_hdr_opt doesn't say sockets must now opt in
via BPF_TCP_OPS_FLAG_*, and the UAPI flags are undocumented.
> diff --git a/include/net/tcp.h b/include/net/tcp.h
> index a12766e38ad4..0f2ceda58319 100644
> --- a/include/net/tcp.h
> +++ b/include/net/tcp.h
> @@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
> {
> if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
> tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
> - bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
> + if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
> + bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
> }
This isn't a bug, but the kernel-doc in struct bpf_tcp_ops no longer matches
what the code does. It still describes rtt() like this:
/* Called on every RTT sample.
* @mrtt: the measured RTT, in microseconds.
* @srtt: the updated smoothed RTT.
*/
void (*rtt)(struct sock *sk, long mrtt, u32 srtt);
but it now runs only when the socket has BPF_TCP_OPS_FLAG_RTT set.
The same applies to the other three callbacks this patch gates:
- parse_hdr() is documented as parsing the options of an incoming skb on an
established connection. It now runs only with
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL set, or with
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN set and rx_opt.saw_unknown true.
- hdr_opt_len() and write_hdr_opt() now run only with
BPF_TCP_OPS_FLAG_WRITE_HDR_OPT set.
None of these comments say that the socket has to opt in, for example by
calling bpf_tcp_ops_set_flags() from the listen() or connect() callback. The
BPF_TCP_OPS_FLAG_* enum in include/uapi/linux/bpf.h has no comments either,
whereas the legacy BPF_SOCK_OPS_*_CB_FLAG values document which callback each
flag enables.
As a result, a program that attaches a bpf_tcp_ops with only .rtt, .parse_hdr,
.hdr_opt_len or .write_hdr_opt, as the struct comments describe, will see
those callbacks never run, with no error. Before this patch they ran
unconditionally.
None of the later patches in the series (up to 3cb5919af2b2) update these
comments.
Could the rtt, parse_hdr, hdr_opt_len and write_hdr_opt comments say which
BPF_TCP_OPS_FLAG_* has to be set on the socket for the callback to run, and
could the UAPI enum values be documented as well?
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/37336989320
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
2026-10-05 15:40 ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Kuniyuki Iwashima
2026-10-05 16:27 ` bot+bpf-ci
@ 2026-10-05 19:10 ` Amery Hung
2026-10-05 19:13 ` Kuniyuki Iwashima
1 sibling, 1 reply; 21+ messages in thread
From: Amery Hung @ 2026-10-05 19:10 UTC (permalink / raw)
To: Kuniyuki Iwashima
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Yonghong Song, John Fastabend, Stanislav Fomichev, Eric Dumazet,
Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 8:45 AM Kuniyuki Iwashima <kuniyu@google.com> wrote:
>
> bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
> incoming / outgoing skb.
>
> bpf_tcp_ops.rtt is called once per RTT, which is every
> incoming skb in ping-pong workloads like tcp_rr.
>
> Even attaching NULL callbacks in the fast path hurts performance.
>
> Let's guard them (and write_hdr_opt) with the new per-socket flags.
>
> Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
> ---
> include/net/tcp.h | 3 ++-
> net/ipv4/tcp_input.c | 7 +++++++
> net/ipv4/tcp_output.c | 20 +++++++++++---------
> 3 files changed, 20 insertions(+), 10 deletions(-)
>
> diff --git a/include/net/tcp.h b/include/net/tcp.h
> index f69091f5e01f..48487135c076 100644
> --- a/include/net/tcp.h
> +++ b/include/net/tcp.h
> @@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
> {90
> if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
> tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
> - bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
> + if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
> + bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
The new per-socket flag load happens before the static-key check
inside bpf_tcp_ops_call(). Thus, even when no bpf_tcp_ops is attached
anywhere, every RTT sample (and other similar places) now touches
tp->bpf_tcp_ops_flags, whereas previously the path stopped at the
disabled static branch.
Can we add something like bpf_tcp_ops_call_flag(op, flag, sk, ...),
with the static-key check first and the flag check before the cgroup
lookup/array scan? This would preserve both the no-attachment fast
path and the flag-clear optimization for attached programs.
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag.
2026-10-05 19:10 ` Amery Hung
@ 2026-10-05 19:13 ` Kuniyuki Iwashima
0 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 19:13 UTC (permalink / raw)
To: Amery Hung
Cc: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi,
Yonghong Song, John Fastabend, Stanislav Fomichev, Eric Dumazet,
Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, bpf, netdev
On Mon, Oct 5, 2026 at 12:10 PM Amery Hung <ameryhung@gmail.com> wrote:
>
> On Mon, Oct 5, 2026 at 8:45 AM Kuniyuki Iwashima <kuniyu@google.com> wrote:
> >
> > bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every
> > incoming / outgoing skb.
> >
> > bpf_tcp_ops.rtt is called once per RTT, which is every
> > incoming skb in ping-pong workloads like tcp_rr.
> >
> > Even attaching NULL callbacks in the fast path hurts performance.
> >
> > Let's guard them (and write_hdr_opt) with the new per-socket flags.
> >
> > Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
> > ---
> > include/net/tcp.h | 3 ++-
> > net/ipv4/tcp_input.c | 7 +++++++
> > net/ipv4/tcp_output.c | 20 +++++++++++---------
> > 3 files changed, 20 insertions(+), 10 deletions(-)
> >
> > diff --git a/include/net/tcp.h b/include/net/tcp.h
> > index f69091f5e01f..48487135c076 100644
> > --- a/include/net/tcp.h
> > +++ b/include/net/tcp.h
> > @@ -3142,7 +3142,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
> > {90
> > if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG))
> > tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt);
> > - bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
> > + if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RTT))
> > + bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
>
> The new per-socket flag load happens before the static-key check
> inside bpf_tcp_ops_call(). Thus, even when no bpf_tcp_ops is attached
> anywhere, every RTT sample (and other similar places) now touches
> tp->bpf_tcp_ops_flags, whereas previously the path stopped at the
> disabled static branch.
>
> Can we add something like bpf_tcp_ops_call_flag(op, flag, sk, ...),
> with the static-key check first and the flag check before the cgroup
> lookup/array scan? This would preserve both the no-attachment fast
> path and the flag-clear optimization for attached programs.
Makes sense, I'll add that in v4.
^ permalink raw reply [flat|nested] 21+ messages in thread
* [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq().
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (3 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 4/9] bpf: tcp: Guard fast-path bpf_tcp_ops_call() under per-socket flag Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
` (3 subsequent siblings)
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
Waking up a thread per packet is expensive when an application
processes variable-length frames (e.g., RPC) that span multiple
packets.
SO_RCVLOWAT can defer wakeups, but because the frame size is
encoded in a fixed-size descriptor at the start of each frame,
the application has to:
1. wake up and recv() the descriptor,
2. raise SO_RCVLOWAT to the payload size via setsockopt(),
3. wake up and recv() the payload, and
4. reset SO_RCVLOWAT back to the descriptor size via
setsockopt() for the next frame.
This requires an extra wakeup and two setsockopt() syscalls
for every single RPC frame.
With SOCKMAP, we can parse skb and suppress wakeups in kernel,
but SOCKMAP adds overhead and also kills zerocopy.
Let's add lighter-weight opt-in callbacks to bpf_tcp_ops to
replace that.
.enqueue_rcvq(): invoked when TCP stack enqueues skb to
sk->sk_receive_queue
.dequeue_rcvq(): invoked in tcp_cleanup_rbuf() after data
is dequeued from sk->sk_receive_queue
Those callbacks can be enabled on a per-socket basis by
bpf_tcp_ops_set_flags():
bpf_tcp_ops_set_flags((struct tcp_sock *)sk,
BPF_TCP_OPS_FLAG_RCVQ, 0);
Later, we will add a new kfunc to adjust sk->sk_rcvlowat from
these callbacks.
This will allow the bpf_tcp_ops prog to parse each skb and
dynamically adjust sk->sk_rcvlowat to suppress unnecessary EPOLLIN
wakeups until sufficient data is available in the receive queue.
The placement of bpf_tcp_ops_call() in tcp_ofo_queue() and
tcp_fastopen_add_skb() is chosen to provide the same snapshot
as tcp_queue_rcv().
For example, if bpf_tcp_ops_call() were called before updating
TCP_SKB_CB(skb)->seq in tcp_fastopen_add_skb(), BPF prog would
need an extra branch for the unlikely TFO case to strip SYN.
In addition, the TCP stack can queue overlapping skbs into recvq.
Once rcv_nxt is updated with a new skb, BPF prog can no longer
infer the previous rcv_nxt from skb->len.
Lastly, dequeue_rcvq() is placed in tcp_cleanup_rbuf() rather
than __tcp_cleanup_rbuf() so that it is not called for sockets
in SOCKMAP, where calling sk->sk_data_ready() from the new
kfunc would otherwise trigger infinite recursion.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
v3: Switch to BPF_TCP_OPS_FLAG_RCVQ
---
include/net/tcp.h | 18 ++++++++++++++++++
include/uapi/linux/bpf.h | 3 ++-
net/ipv4/bpf_tcp_ops.c | 10 ++++++++++
net/ipv4/tcp.c | 2 ++
net/ipv4/tcp_fastopen.c | 2 ++
net/ipv4/tcp_input.c | 4 ++++
tools/include/uapi/linux/bpf.h | 3 ++-
7 files changed, 40 insertions(+), 2 deletions(-)
diff --git a/include/net/tcp.h b/include/net/tcp.h
index 48487135c076..7f4ab50f08c7 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -3054,6 +3054,12 @@ struct bpf_tcp_ops {
struct request_sock *req, struct sk_buff *syn_skb,
enum tcp_synack_type synack_type,
u32 opt_off);
+
+ /* Called when an incoming skb is enqueued to sk->sk_receive_queue. */
+ void (*enqueue_rcvq)(struct sock *sk, struct sk_buff *skb);
+
+ /* Called after data is dequeued from sk->sk_receive_queue. */
+ void (*dequeue_rcvq)(struct sock *sk);
};
#define bpf_tcp_ops_call(op, sk, ...) \
@@ -3146,6 +3152,18 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt)
bpf_tcp_ops_call(rtt, sk, mrtt, srtt);
}
+static inline void bpf_tcp_ops_enqueue_rcvq(struct sock *sk, struct sk_buff *skb)
+{
+ if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RCVQ))
+ bpf_tcp_ops_call(enqueue_rcvq, sk, skb);
+}
+
+static inline void bpf_tcp_ops_dequeue_rcvq(struct sock *sk)
+{
+ if (BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), RCVQ))
+ bpf_tcp_ops_call(dequeue_rcvq, sk);
+}
+
#if IS_ENABLED(CONFIG_SMC)
extern struct static_key_false tcp_have_smc;
#endif
diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
index 6ff90b73dd38..ded3ac8df9ac 100644
--- a/include/uapi/linux/bpf.h
+++ b/include/uapi/linux/bpf.h
@@ -7339,7 +7339,8 @@ enum {
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
- BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
+ BPF_TCP_OPS_FLAG_RCVQ = (1 << 4),
+ BPF_TCP_OPS_FLAG_ALL = (1 << 5) - 1,
};
/* List of TCP states. There is a build check in net/ipv4/tcp.c to detect
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index 5693857e764e..dafe2337f9fc 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -76,6 +76,14 @@ static void write_hdr_opt_stub(struct sock *sk, struct sk_buff *skb,
{
}
+static void enqueue_rcvq_stub(struct sock *sk, struct sk_buff *skb)
+{
+}
+
+static void dequeue_rcvq_stub(struct sock *sk)
+{
+}
+
static struct bpf_tcp_ops __bpf_tcp_ops = {
.timeout_init = timeout_init_stub,
.rwnd_init = rwnd_init_stub,
@@ -90,6 +98,8 @@ static struct bpf_tcp_ops __bpf_tcp_ops = {
.parse_hdr = parse_hdr_stub,
.hdr_opt_len = hdr_opt_len_stub,
.write_hdr_opt = write_hdr_opt_stub,
+ .enqueue_rcvq = enqueue_rcvq_stub,
+ .dequeue_rcvq = dequeue_rcvq_stub,
};
BPF_CALL_4(bpf_tcp_ops_store_hdr_opt, void *, ctx, const void *, from,
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
index fa69961c47d3..a1e2bb3974ce 100644
--- a/net/ipv4/tcp.c
+++ b/net/ipv4/tcp.c
@@ -1609,6 +1609,8 @@ void tcp_cleanup_rbuf(struct sock *sk, int copied)
"cleanup rbuf bug: copied %X seq %X rcvnxt %X\n",
tp->copied_seq, TCP_SKB_CB(skb)->end_seq, tp->rcv_nxt);
__tcp_cleanup_rbuf(sk, copied);
+
+ bpf_tcp_ops_dequeue_rcvq(sk);
}
static void tcp_eat_recv_skb(struct sock *sk, struct sk_buff *skb)
diff --git a/net/ipv4/tcp_fastopen.c b/net/ipv4/tcp_fastopen.c
index 471c78be5513..4939bcbc81d1 100644
--- a/net/ipv4/tcp_fastopen.c
+++ b/net/ipv4/tcp_fastopen.c
@@ -281,6 +281,8 @@ void tcp_fastopen_add_skb(struct sock *sk, struct sk_buff *skb)
TCP_SKB_CB(skb)->seq++;
TCP_SKB_CB(skb)->tcp_flags &= ~TCPHDR_SYN;
+ bpf_tcp_ops_enqueue_rcvq(sk, skb);
+
tp->rcv_nxt = TCP_SKB_CB(skb)->end_seq;
tcp_add_receive_queue(sk, skb);
tp->syn_data_acked = 1;
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 79d721215f52..8518c744aa17 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -5351,6 +5351,8 @@ static void tcp_ofo_queue(struct sock *sk)
continue;
}
+ bpf_tcp_ops_enqueue_rcvq(sk, skb);
+
tail = skb_peek_tail(&sk->sk_receive_queue);
eaten = tail && tcp_try_coalesce(sk, tail, skb, &fragstolen);
tcp_rcv_nxt_update(tp, TCP_SKB_CB(skb)->end_seq);
@@ -5554,6 +5556,8 @@ static int __must_check tcp_queue_rcv(struct sock *sk, struct sk_buff *skb,
int eaten;
struct sk_buff *tail = skb_peek_tail(&sk->sk_receive_queue);
+ bpf_tcp_ops_enqueue_rcvq(sk, skb);
+
eaten = (tail &&
tcp_try_coalesce(sk, tail,
skb, fragstolen)) ? 1 : 0;
diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h
index 6ff90b73dd38..ded3ac8df9ac 100644
--- a/tools/include/uapi/linux/bpf.h
+++ b/tools/include/uapi/linux/bpf.h
@@ -7339,7 +7339,8 @@ enum {
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_ALL = (1 << 1),
BPF_TCP_OPS_FLAG_PARSE_HDR_OPT_UNKNOWN = (1 << 2),
BPF_TCP_OPS_FLAG_WRITE_HDR_OPT = (1 << 3),
- BPF_TCP_OPS_FLAG_ALL = (1 << 4) - 1,
+ BPF_TCP_OPS_FLAG_RCVQ = (1 << 4),
+ BPF_TCP_OPS_FLAG_ALL = (1 << 5) - 1,
};
/* List of TCP states. There is a build check in net/ipv4/tcp.c to detect
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat().
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (4 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 5/9] bpf: tcp: Introduce bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
` (2 subsequent siblings)
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev, Emil Tsalapatis
We will add a kfunc for bpf_tcp_ops.{enqueue,dequeue}_rcvq()
to adjust sk->sk_rcvlowat.
These hooks are triggered
* when the TCP stack enqueues an skb to sk->sk_receive_queue
* after data is dequeued from sk->sk_receive_queue
In the enqueue path, tcp_data_ready() is always called after
the hooks in tcp_queue_rcv() and tcp_ofo_queue().
If tcp_set_rcvlowat() were used as is, tcp_data_ready() could
be called twice for the same skb, which is redundant and also
confusing.
Let's split out __tcp_set_rcvlowat() and add a flag to control
wakeup behaviour.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
Acked-by: Stanislav Fomichev <sdf@fomichev.me>
Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com>
---
include/net/tcp.h | 1 +
net/ipv4/tcp.c | 12 +++++++++---
2 files changed, 10 insertions(+), 3 deletions(-)
diff --git a/include/net/tcp.h b/include/net/tcp.h
index 7f4ab50f08c7..adc8d1a5f926 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -512,6 +512,7 @@ void tcp_set_keepalive(struct sock *sk, int val);
void tcp_syn_ack_timeout(const struct request_sock *req);
int tcp_recvmsg(struct sock *sk, struct msghdr *msg, size_t len,
int flags);
+int __tcp_set_rcvlowat(struct sock *sk, int val, bool wakeup);
int tcp_set_rcvlowat(struct sock *sk, int val);
void tcp_set_rcvbuf(struct sock *sk, int val);
int tcp_set_window_clamp(struct sock *sk, int val);
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
index a1e2bb3974ce..ce35060f82ee 100644
--- a/net/ipv4/tcp.c
+++ b/net/ipv4/tcp.c
@@ -1827,8 +1827,7 @@ int tcp_peek_len(struct socket *sock)
return tcp_inq(sock->sk);
}
-/* Make sure sk_rcvbuf is big enough to satisfy SO_RCVLOWAT hint */
-int tcp_set_rcvlowat(struct sock *sk, int val)
+int __tcp_set_rcvlowat(struct sock *sk, int val, bool wakeup)
{
struct tcp_sock *tp = tcp_sk(sk);
int space, cap;
@@ -1841,7 +1840,8 @@ int tcp_set_rcvlowat(struct sock *sk, int val)
WRITE_ONCE(sk->sk_rcvlowat, val ? : 1);
/* Check if we need to signal EPOLLIN right now */
- tcp_data_ready(sk);
+ if (wakeup)
+ tcp_data_ready(sk);
if (sk->sk_userlocks & SOCK_RCVBUF_LOCK)
return 0;
@@ -1856,6 +1856,12 @@ int tcp_set_rcvlowat(struct sock *sk, int val)
return 0;
}
+/* Make sure sk_rcvbuf is big enough to satisfy SO_RCVLOWAT hint */
+int tcp_set_rcvlowat(struct sock *sk, int val)
+{
+ return __tcp_set_rcvlowat(sk, val, true);
+}
+
void tcp_set_rcvbuf(struct sock *sk, int val)
{
tcp_set_window_clamp(sk, tcp_win_from_space(sk, val));
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (5 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 6/9] tcp: Split out __tcp_set_rcvlowat() Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
The next patch exposes a new kfunc calling __tcp_set_rcvlowat()
to bpf_tcp_ops.
MPTCP has its own sock->ops->set_rcvlowat() / mptcp_set_rcvlowat(),
so we should not allow calling __tcp_set_rcvlowat() on MPTCP
subflows.
Let's disable BPF_TCP_OPS_FLAG_RCVQ for MPTCP for now.
If needed in the future, bpf_tcp_ops_set_rcvlowat() could be
extended to properly support MPTCP.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
net/ipv4/bpf_tcp_ops.c | 3 +++
1 file changed, 3 insertions(+)
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index dafe2337f9fc..c73d3478a8d5 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -347,6 +347,9 @@ __bpf_kfunc int bpf_tcp_ops_set_flags(struct tcp_sock *tp, u32 enable, u32 disab
if ((enable & disable) || (enable | disable) & ~BPF_TCP_OPS_FLAG_ALL)
return -EINVAL;
+ if (sk_is_mptcp((struct sock *)tp) && (enable & BPF_TCP_OPS_FLAG_RCVQ))
+ return -EOPNOTSUPP;
+
old = READ_ONCE(tp->bpf_tcp_ops_flags);
do {
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat.
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (6 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 7/9] bpf: mptcp: Don't support BPF_TCP_OPS_FLAG_RCVQ Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
2026-10-05 15:40 ` [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq() Kuniyuki Iwashima
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev, Emil Tsalapatis
bpf_tcp_ops.{enqueue,dequeue}_rcvq() were added to parse skb and
adjust sk->sk_rcvlowat dynamically to suppress unnecessary wakeups.
Let's add a new kfunc to set sk->sk_rcvlowat.
Negative values are clamped to INT_MAX, consistent with SO_RCVLOWAT.
For enqueue_rcvq(), wakeup is set to false because:
* tcp_data_ready() is always called after the hooks in
tcp_queue_rcv() and tcp_ofo_queue().
* when tcp_fastopen_add_skb() is called for TFO SYN, the socket is
not yet accept()ed, and when called for TFO SYN+ACK, the socket
is woken up by sk->sk_state_change() anyway.
For dequeue_rcvq(), wakeup is set to true because tcp_data_ready()
is not called in that path.
An alternative would be to support bpf_setsockopt() for these
hooks.
However, that approach involves excessive conditionals and an
unnecessary memcpy(), costs we do not want to pay for every skb
in the TCP fast path.
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
Acked-by: Stanislav Fomichev <sdf@fomichev.me>
Tested-by: Clément Léger <cleger@meta.com>
Reviewed-by: Emil Tsalapatis <emil@etsalapatis.com>
---
net/ipv4/bpf_tcp_ops.c | 46 ++++++++++++++++++++++++++++++++++++++++++
1 file changed, 46 insertions(+)
diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c
index c73d3478a8d5..4e4846121ce7 100644
--- a/net/ipv4/bpf_tcp_ops.c
+++ b/net/ipv4/bpf_tcp_ops.c
@@ -361,12 +361,31 @@ __bpf_kfunc int bpf_tcp_ops_set_flags(struct tcp_sock *tp, u32 enable, u32 disab
return 0;
}
+__bpf_kfunc int bpf_tcp_ops_set_rcvlowat(struct sock *sk, int rcvlowat,
+ const struct bpf_prog_aux *aux)
+{
+ u32 moff = aux->attach_st_ops_member_off;
+ bool wakeup = false;
+
+ if (moff == offsetof(struct bpf_tcp_ops, dequeue_rcvq))
+ wakeup = true;
+
+ if (rcvlowat < 0)
+ rcvlowat = INT_MAX;
+
+ return __tcp_set_rcvlowat(sk, rcvlowat, wakeup);
+}
+
__bpf_kfunc_end_defs();
BTF_KFUNCS_START(bpf_tcp_ops_set_flags_kfunc_set)
BTF_ID_FLAGS(func, bpf_tcp_ops_set_flags)
BTF_KFUNCS_END(bpf_tcp_ops_set_flags_kfunc_set)
+BTF_KFUNCS_START(bpf_tcp_ops_set_rcvlowat_kfunc_set)
+BTF_ID_FLAGS(func, bpf_tcp_ops_set_rcvlowat, KF_IMPLICIT_ARGS)
+BTF_KFUNCS_END(bpf_tcp_ops_set_rcvlowat_kfunc_set)
+
static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog,
u32 kfunc_id)
{
@@ -386,6 +405,31 @@ static const struct btf_kfunc_id_set bpf_tcp_ops_set_flags_kfunc_id_set = {
.filter = bpf_tcp_ops_set_flags_kfunc_filter,
};
+static int bpf_tcp_ops_set_rcvlowat_kfunc_filter(const struct bpf_prog *prog,
+ u32 kfunc_id)
+{
+ u32 moff;
+
+ if (!btf_id_set8_contains(&bpf_tcp_ops_set_rcvlowat_kfunc_set, kfunc_id))
+ return 0;
+
+ if (prog->aux->st_ops != &bpf_tcp_ops)
+ return -EACCES;
+
+ moff = prog->aux->attach_st_ops_member_off;
+ if (moff != offsetof(struct bpf_tcp_ops, enqueue_rcvq) &&
+ moff != offsetof(struct bpf_tcp_ops, dequeue_rcvq))
+ return -EACCES;
+
+ return 0;
+}
+
+static const struct btf_kfunc_id_set bpf_tcp_ops_set_rcvlowat_kfunc_id_set = {
+ .owner = THIS_MODULE,
+ .set = &bpf_tcp_ops_set_rcvlowat_kfunc_set,
+ .filter = bpf_tcp_ops_set_rcvlowat_kfunc_filter,
+};
+
static int __init __bpf_tcp_ops_init(void)
{
int ret;
@@ -394,6 +438,8 @@ static int __init __bpf_tcp_ops_init(void)
&bpf_tcp_ops_set_flags_kfunc_id_set);
ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT,
&bpf_tcp_ops_set_flags_kfunc_id_set);
+ ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
+ &bpf_tcp_ops_set_rcvlowat_kfunc_id_set);
ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops);
return ret;
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread* [PATCH v3 bpf-next 9/9] selftest: bpf: Add test for bpf_tcp_ops.{enqueue,dequeue}_rcvq().
2026-10-05 15:40 [PATCH v3 bpf-next 0/9] bpf: Add bpf_tcp_ops hooks for TCP AutoLOWAT Kuniyuki Iwashima
` (7 preceding siblings ...)
2026-10-05 15:40 ` [PATCH v3 bpf-next 8/9] bpf: tcp: Add kfunc to adjust sk->sk_rcvlowat Kuniyuki Iwashima
@ 2026-10-05 15:40 ` Kuniyuki Iwashima
8 siblings, 0 replies; 21+ messages in thread
From: Kuniyuki Iwashima @ 2026-10-05 15:40 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Martin KaFai Lau, Eduard Zingerman, Kumar Kartikeya Dwivedi
Cc: Amery Hung, Yonghong Song, John Fastabend, Stanislav Fomichev,
Eric Dumazet, Neal Cardwell, Willem de Bruijn, Tenzin Ukyab,
Clément Léger, Kuniyuki Iwashima, Kuniyuki Iwashima,
bpf, netdev
The test is roughly divided into two stages, and the sequence
is as follows:
I) Setup
1. Attach two BPF programs to a cgroup
2. Establish a TCP connection (@client <-> @child) within the cgroup
3. Enable BPF_TCP_OPS_FLAG_RCVQ on @child via setsockopt()
II) RPC frame exchange in various patterns
4. Send a partial RPC descriptor from @client to @child
5. Verify that epoll does NOT wake up @child
6. Send the remaining data of the RPC frame
7. Verify that epoll finally wakes up @child
During setup, two BPF programs are attached to simulate
a real-world scenario; one is bpf_tcp_ops and the other is
CGROUP_SOCKOPT.
While the bpf_tcp_ops prog handles the dynamic adjustment of
sk->sk_rcvlowat, the CGROUP_SOCKOPT prog is used to enable
the TCP AutoLOWAT feature via userspace setsockopt() using
pseudo options:
#define SOL_BPF 0xdeadbeef
#define BPF_TCP_AUTOLOWAT 0x8badf00d
setsockopt(fd, SOL_BPF, BPF_TCP_AUTOLOWAT, &(int){1}, sizeof(int));
This reflects a common production use case where an application
decides to start parsing RPC frames only at a certain point in
the stream (e.g., after HTTP Upgrade), rather than immediately
after TCP 3WHS (->passive_established(), etc).
When BPF_TCP_AUTOLOWAT is enabled, the BPF prog sets
BPF_TCP_OPS_FLAG_RCVQ and initializes sk_local_storage
for two sequence numbers to manage its state.
Then, for the RPC frame exchange, this test uses a simple format
defined as follows:
0 8 16 24 32
+--------+--------+-------+--------+ `.
| header size | |
+--------+--------+-------+--------+ > RPC descriptor (8 bytes)
| payload size | |
+--------+--------+-------+--------+ .'
~ header ~
+--------+--------+-------+--------+
~ payload ~
+--------+--------+-------+--------+
Every time a new skb is enqueued to sk->sk_receive_queue, the
bpf_tcp_ops prog parses it and updates these sequence numbers:
rpc_desc_seq : the SEQ # of the start of the RPC descriptor
rpc_end_seq : the SEQ # of the end of the RPC frame
=> rpc_desc_seq + 8 + header size + payload size
Assume we receive two RPC descriptors in the following pattern:
1. When we receive skb-1, only part of the RPC descriptor is parsed.
rpc_desc_seq is set to the first byte while rpc_end_seq is
unknown. Thus, sk->sk_rcvlowat is set to the size of the RPC
descriptor (8 bytes).
<- skb-1 -> <---- skb-2 ----> <------ skb-3 ----->
+-----------+.................+....................+......
| RPC desc 1 | header + payload | RPC desc 2 | ...
+-----------+.................+....................+......
^ ^-.
`- rpc_desc_seq `- sk->sk_rcvlowat
2. Next, we receive skb-2, which completes the first RPC descriptor.
Now rpc_end_seq is known, so sk->sk_rcvlowat is advanced to it.
<- skb-1 -> <---- skb-2 ----> <------ skb-3 ----->
+-----------+-----------------+....................+......
| RPC desc 1 | header + payload | RPC desc 2 | ...
+-----------+-----------------+....................+......
^ ^
'- rpc_desc_seq '- rpc_end_seq
& sk->sk_rcvlowat
3. Once we receive skb-3, which contains the next full RPC descriptor,
rpc_desc_seq is advanced and rpc_end_seq is updated according
to the size of RPC frame 2.
Note that sk->sk_rcvlowat is NOT updated to the new rpc_end_seq
yet. This ensures that the application is woken up to read the
already complete RPC frame 1.
<- skb-1 -> <---- skb-2 ----> <------ skb-3 ----->
+-----------+-----------------+--------------------+......
| RPC desc 1 | header + payload | RPC desc 2 | ... |
+-----------+-----------------+--------------------+......
^ ^
rpc_desc_seq -----------' rpc_end_seq ----...-'
& sk->sk_rcvlowat
This sequence corresponds to the 4th test case in rpc_test_cases[],
and we can see helpful output if we "#define DEBUG":
# cat /sys/kernel/tracing/trace_pipe | \
awk '{ if ($0 ~ /AF_/) sub(/^.*AF_/, "AF_"); print $0 }' & \
BGPID=$!; ./test_progs -t tcp_autolowat; kill -9 -$BGPID
...
AF_INET6 rpc_test_cases[3]: Start parsing skb: seq: 0, end_seq: 1, len: 1, rpc_desc_seq: 0, rpc_end_seq: 0, rpc_buff_len: 0
AF_INET6 rpc_test_cases[3]: Copied 1 bytes: rpc_desc_buff_len: 1
AF_INET6 rpc_test_cases[3]: Setting rcvlowat: tp->copied_seq: 0, rpc_desc_seq: 0, rpc_end_seq: 0, rpc_desc_buff_len: 1
AF_INET6 rpc_test_cases[3]: Set rcvlowat: expected: 8, actual: 8
AF_INET6 rpc_test_cases[3]: Start parsing skb: seq: 1, end_seq: 8, len: 7, rpc_desc_seq: 0, rpc_end_seq: 0, rpc_buff_len: 1
AF_INET6 rpc_test_cases[3]: Copied full descriptor: rpc_desc_seq: 0, rpc_end_seq: 258, header_len: 100, payload_len: 150
AF_INET6 rpc_test_cases[3]: No more descriptor: rpc_end_seq: 258, end_seq: 8
AF_INET6 rpc_test_cases[3]: Setting rcvlowat: tp->copied_seq: 0, rpc_desc_seq: 0, rpc_end_seq: 258, rpc_desc_buff_len: 8
AF_INET6 rpc_test_cases[3]: Set rcvlowat: expected: 258, actual: 258
...
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
v3:
* Use bpf_tcp_ops_set_flags()
v2:
* Avoid address comparison for a specific version of gcc.
* Make rpc_test_cases[] static.
* Update comment in rpc_test_case[].
---
.../selftests/bpf/prog_tests/tcp_autolowat.c | 350 ++++++++++++++++++
.../selftests/bpf/progs/bpf_tracing_net.h | 2 +
.../selftests/bpf/progs/tcp_autolowat.c | 294 +++++++++++++++
3 files changed, 646 insertions(+)
create mode 100644 tools/testing/selftests/bpf/prog_tests/tcp_autolowat.c
create mode 100644 tools/testing/selftests/bpf/progs/tcp_autolowat.c
diff --git a/tools/testing/selftests/bpf/prog_tests/tcp_autolowat.c b/tools/testing/selftests/bpf/prog_tests/tcp_autolowat.c
new file mode 100644
index 000000000000..337f9d34a39c
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/tcp_autolowat.c
@@ -0,0 +1,350 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright 2026 Google LLC */
+#include <sys/epoll.h>
+
+#include "test_progs.h"
+#include "cgroup_helpers.h"
+#include "network_helpers.h"
+
+#include "tcp_autolowat.skel.h"
+
+#define SOL_BPF 0xdeadbeef
+#define BPF_TCP_AUTOLOWAT 0x8badf00d
+
+struct rpc_descriptor {
+ u32 header_len;
+ u32 payload_len;
+};
+
+enum rpc_event_type {
+ RPC_EVENT_END,
+ RPC_EVENT_AUTOLOWAT,
+ RPC_EVENT_SEND,
+ RPC_EVENT_RECV,
+ RPC_EVENT_EPOLL,
+ RPC_EVENT_RCVLOWAT,
+};
+
+struct rpc_event {
+ enum rpc_event_type type;
+ union {
+ int len;
+ int nfds;
+ int val;
+ int rcvlowat;
+ };
+};
+
+#define RPC_DESC_SIZE (sizeof(struct rpc_descriptor))
+
+static struct rpc_test_case {
+ char data[4096];
+ struct rpc_descriptor desc[32];
+ struct rpc_event event[32];
+} rpc_test_cases[] = {
+ {
+ .desc = {
+ { .header_len = 100, .payload_len = 150 },
+ },
+ .event = {
+ { .type = RPC_EVENT_AUTOLOWAT, .val = 1},
+ /* Single full RPC message in skb. */
+ { .type = RPC_EVENT_SEND, .len = RPC_DESC_SIZE + 100 + 150},
+ { .type = RPC_EVENT_EPOLL, .nfds = 1},
+ { .type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE + 100 + 150},
+ },
+ },
+ {
+ .desc = {
+ {.header_len = 100, .payload_len = 150},
+ {.header_len = 100, .payload_len = 150},
+ {.header_len = 100, .payload_len = 150},
+ },
+ .event = {
+ { .type = RPC_EVENT_AUTOLOWAT, .val = 1},
+ /* Two full RPC messages in skb. */
+ {.type = RPC_EVENT_SEND, .len = (RPC_DESC_SIZE + 100 + 150) * 2},
+ {.type = RPC_EVENT_EPOLL, .nfds = 1},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = (RPC_DESC_SIZE + 100 + 150) * 2},
+ /* Single full RPC message in skb. */
+ { .type = RPC_EVENT_SEND, .len = RPC_DESC_SIZE + 100 + 150},
+ { .type = RPC_EVENT_EPOLL, .nfds = 1},
+ { .type = RPC_EVENT_RCVLOWAT, .rcvlowat = (RPC_DESC_SIZE + 100 + 150) * 3},
+ },
+ },
+ {
+ .desc = {
+ {.header_len = 100, .payload_len = 150},
+ {.header_len = 100, .payload_len = 150},
+ {.header_len = 100, .payload_len = 150},
+ },
+ .event = {
+ { .type = RPC_EVENT_AUTOLOWAT, .val = 1},
+ /* Two full RPC messages in skb. */
+ {.type = RPC_EVENT_SEND, .len = (RPC_DESC_SIZE + 100 + 150) * 2},
+ {.type = RPC_EVENT_EPOLL, .nfds = 1},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = (RPC_DESC_SIZE + 100 + 150) * 2},
+ /* Only the next descriptor in skb. */
+ { .type = RPC_EVENT_SEND, .len = RPC_DESC_SIZE},
+ { .type = RPC_EVENT_EPOLL, .nfds = 1},
+ { .type = RPC_EVENT_RCVLOWAT, .rcvlowat = (RPC_DESC_SIZE + 100 + 150) * 2},
+ },
+ },
+ {
+ .desc = {
+ {.header_len = 100, .payload_len = 150},
+ {.header_len = 200, .payload_len = 500},
+ },
+ .event = {
+ { .type = RPC_EVENT_AUTOLOWAT, .val = 1},
+ /* The first descriptor is partial. */
+ {.type = RPC_EVENT_SEND, .len = 1},
+ {.type = RPC_EVENT_EPOLL, .nfds = 0},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE},
+ /* The first descriptor is available. */
+ {.type = RPC_EVENT_SEND, .len = RPC_DESC_SIZE - 1},
+ {.type = RPC_EVENT_EPOLL, .nfds = 0},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE + 150 + 100},
+ /* The first header is ready. */
+ {.type = RPC_EVENT_SEND, .len = 100},
+ {.type = RPC_EVENT_EPOLL, .nfds = 0},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE + 150 + 100},
+ /* skb has the first payload and 1 byte of the next descriptor. */
+ {.type = RPC_EVENT_SEND, .len = 150 + 1},
+ {.type = RPC_EVENT_EPOLL, .nfds = 1},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE + 150 + 100},
+ /* After reading the first RPC message, SO_RCVLOWAT should be RPC_DESC_SIZE. */
+ {.type = RPC_EVENT_RECV, .len = RPC_DESC_SIZE + 150 + 100},
+ {.type = RPC_EVENT_EPOLL, .nfds = 0},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE},
+ /* The second descriptor is available. */
+ {.type = RPC_EVENT_SEND, .len = RPC_DESC_SIZE - 1},
+ {.type = RPC_EVENT_EPOLL, .nfds = 0},
+ {.type = RPC_EVENT_RCVLOWAT, .rcvlowat = RPC_DESC_SIZE + 200 + 500},
+ },
+ },
+};
+
+struct tcp_autolowat_test_cb {
+ int saved_netns;
+ union {
+ int fd[4];
+ struct {
+ int server, client, child;
+ int epoll;
+ };
+ };
+};
+
+static void tcp_autolowat_teardown_cb(struct tcp_autolowat_test_cb *cb)
+{
+ int i, err;
+
+ for (i = 0; i < ARRAY_SIZE(cb->fd); i++) {
+ if (cb->fd[i] != -1)
+ close(cb->fd[i]);
+ }
+
+ if (cb->saved_netns != -1) {
+ err = setns(cb->saved_netns, CLONE_NEWNET);
+ ASSERT_OK(err, "restore netns");
+
+ close(cb->saved_netns);
+ }
+}
+
+static int tcp_autolowat_setup_cb(struct tcp_autolowat_test_cb *cb, int family)
+{
+ struct epoll_event ev = {};
+ int err;
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(cb->fd); i++)
+ cb->fd[i] = -1;
+
+ cb->saved_netns = open("/proc/self/ns/net", O_RDONLY);
+ if (!ASSERT_OK_FD(cb->saved_netns, "save netns"))
+ goto err;
+
+ err = unshare(CLONE_NEWNET);
+ if (!ASSERT_OK(err, "unshare"))
+ goto err;
+
+ err = system("ip link set dev lo up");
+ if (!ASSERT_OK(err, "set up lo"))
+ goto err;
+
+ cb->server = start_server(family, SOCK_STREAM, NULL, 0, 0);
+ if (!ASSERT_OK_FD(cb->server, "start_server"))
+ goto err;
+
+ cb->client = connect_to_fd(cb->server, 0);
+ if (!ASSERT_OK_FD(cb->client, "connect_to_fd"))
+ goto err;
+
+ cb->child = accept(cb->server, NULL, NULL);
+ if (!ASSERT_OK_FD(cb->child, "accept"))
+ goto err;
+
+ cb->epoll = epoll_create1(0);
+ if (!ASSERT_OK_FD(cb->epoll, "epoll_create"))
+ goto err;
+
+ ev.events = EPOLLIN;
+ ev.data.fd = cb->child;
+
+ err = epoll_ctl(cb->epoll, EPOLL_CTL_ADD, cb->child, &ev);
+ if (!ASSERT_OK(err, "epoll_ctl"))
+ goto err;
+
+ return 0;
+
+err:
+ tcp_autolowat_teardown_cb(cb);
+ return -1;
+}
+
+static int tcp_autolowat_build_data(struct rpc_test_case *test_case)
+{
+ struct rpc_descriptor *desc = test_case->desc;
+ char *ptr = test_case->data;
+ int rpc_size;
+
+ memset(ptr, 0, sizeof(test_case->data));
+
+ while (desc->header_len + desc->payload_len) {
+ rpc_size = sizeof(*desc) + desc->header_len + desc->payload_len;
+
+ if (!ASSERT_LE(ptr + rpc_size - test_case->data,
+ sizeof(test_case->data), "data overflow"))
+ return 1;
+
+ memcpy(ptr, desc, sizeof(*desc));
+ ptr += rpc_size;
+ desc++;
+ }
+
+ if (!ASSERT_GT(ptr - test_case->data, 0, "no data"))
+ return 1;
+
+ return 0;
+}
+
+static void tcp_autolowat_run_rpc_test(struct tcp_autolowat_test_cb *cb,
+ struct rpc_test_case *test_case)
+{
+ struct rpc_event *event = test_case->event;
+ char *ptr = test_case->data;
+ struct epoll_event ev;
+ socklen_t optlen;
+ int err, optval;
+ char buf[4096];
+
+ if (tcp_autolowat_build_data(test_case))
+ return;
+
+ while (1) {
+ switch (event->type) {
+ case RPC_EVENT_END:
+ return;
+ case RPC_EVENT_AUTOLOWAT:
+ err = setsockopt(cb->child, SOL_BPF, BPF_TCP_AUTOLOWAT,
+ &event->val, sizeof(event->val));
+ if (!ASSERT_OK(err, "setsockopt"))
+ return;
+ break;
+ case RPC_EVENT_SEND:
+ err = send(cb->client, ptr, event->len, 0);
+ if (!ASSERT_EQ(err, event->len, "send"))
+ return;
+
+ ptr += event->len;
+ break;
+ case RPC_EVENT_RECV:
+ err = recv(cb->child, buf, event->len, 0);
+ if (!ASSERT_EQ(err, event->len, "recv"))
+ return;
+ break;
+ case RPC_EVENT_EPOLL:
+ err = epoll_wait(cb->epoll, &ev, 1, 100);
+ if (!ASSERT_EQ(err, event->nfds, "epoll_wait"))
+ return;
+ break;
+ case RPC_EVENT_RCVLOWAT:
+ optval = 0;
+ optlen = sizeof(optval);
+
+ err = getsockopt(cb->child, SOL_SOCKET, SO_RCVLOWAT, &optval, &optlen);
+ if (!ASSERT_OK(err, "getsockopt") ||
+ !ASSERT_EQ(optval, event->rcvlowat, "rcvlowat"))
+ return;
+ break;
+ }
+
+ event++;
+ }
+}
+
+static void tcp_autolowat_run_rpc_tests(struct tcp_autolowat *skel, int family)
+{
+ struct tcp_autolowat_test_cb cb;
+ int err;
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(rpc_test_cases); i++) {
+ memset(skel->bss->test_name, 0, sizeof(skel->bss->test_name));
+
+ snprintf(skel->bss->test_name, sizeof(skel->bss->test_name),
+ "AF_INET%c rpc_test_cases[%d]",
+ family == AF_INET ? ' ' : '6', i);
+
+ if (!test__start_subtest(skel->bss->test_name))
+ continue;
+
+ err = tcp_autolowat_setup_cb(&cb, family);
+ if (err)
+ continue;
+
+ tcp_autolowat_run_rpc_test(&cb, &rpc_test_cases[i]);
+ tcp_autolowat_teardown_cb(&cb);
+ }
+}
+
+static void tcp_autolowat_run_tests(struct tcp_autolowat *skel)
+{
+ tcp_autolowat_run_rpc_tests(skel, AF_INET);
+ tcp_autolowat_run_rpc_tests(skel, AF_INET6);
+}
+
+void test_tcp_autolowat(void)
+{
+ struct tcp_autolowat *skel;
+ struct bpf_link *link[2];
+ int cgroup;
+
+ skel = tcp_autolowat__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ return;
+
+ cgroup = test__join_cgroup("/tcp_autolowat");
+ if (!ASSERT_GE(cgroup, 0, "join_cgroup"))
+ goto destroy_skel;
+
+ link[0] = bpf_map__attach_cgroup_opts(skel->maps.tcp_autolowat_ops, cgroup, NULL);
+ if (!ASSERT_OK_PTR(link[0], "attach_cgroup(tcp_autolowat_ops)"))
+ goto close_cgroup;
+
+ link[1] = bpf_program__attach_cgroup(skel->progs.tcp_autolowat_setsockopt, cgroup);
+ if (!ASSERT_OK_PTR(link[1], "attach_cgroup(SETSOCKOPT)"))
+ goto destroy_sockops;
+
+ tcp_autolowat_run_tests(skel);
+
+ bpf_link__destroy(link[1]);
+destroy_sockops:
+ bpf_link__destroy(link[0]);
+close_cgroup:
+ close(cgroup);
+destroy_skel:
+ tcp_autolowat__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/progs/bpf_tracing_net.h b/tools/testing/selftests/bpf/progs/bpf_tracing_net.h
index 593b38f90417..4c999d59cbbc 100644
--- a/tools/testing/selftests/bpf/progs/bpf_tracing_net.h
+++ b/tools/testing/selftests/bpf/progs/bpf_tracing_net.h
@@ -79,6 +79,8 @@
#define NEXTHDR_TCP 6
+#define TCPHDR_FIN 0x01
+
#define TCPOPT_NOP 1
#define TCPOPT_EOL 0
#define TCPOPT_MSS 2
diff --git a/tools/testing/selftests/bpf/progs/tcp_autolowat.c b/tools/testing/selftests/bpf/progs/tcp_autolowat.c
new file mode 100644
index 000000000000..8a55f3cee260
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/tcp_autolowat.c
@@ -0,0 +1,294 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright 2026 Google LLC */
+#include "vmlinux.h"
+
+#include <string.h>
+#include <limits.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+#include <bpf/bpf_core_read.h>
+
+#include "bpf_tracing_net.h"
+
+#define SOL_BPF 0xdeadbeef
+#define BPF_TCP_AUTOLOWAT 0x8badf00d
+
+//#define DEBUG /* For verbose output. */
+
+struct rpc_descriptor {
+ u32 header_len;
+ u32 payload_len;
+};
+
+#define RPC_DESC_SIZE (sizeof(struct rpc_descriptor))
+#define MAX_RPC_DESC_PER_SKB 100
+
+struct tcp_autolowat_cb {
+ /* Don't put this field at the end; BPF verifier complains. */
+ char rpc_desc_buf[RPC_DESC_SIZE];
+ u32 rpc_desc_seq;
+ u32 rpc_end_seq;
+#ifdef DEBUG
+ u32 isn;
+#endif
+ u8 rpc_desc_buff_len;
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_SK_STORAGE);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __type(key, int);
+ __type(value, struct tcp_autolowat_cb);
+} tcp_autolowat_map SEC(".maps");
+
+char test_name[64];
+
+#ifdef DEBUG
+#define LOG(str, ...) \
+ bpf_printk("%s: " str, test_name, ##__VA_ARGS__)
+#else
+#define LOG(...)
+#endif
+
+#define SEQ(val) \
+ (val - cb->isn)
+#define TP_SEQ(field) \
+ (tp->field - cb->isn)
+#define CB_SEQ(field) \
+ (cb->field - cb->isn)
+
+static int tcp_parse_descriptor(struct tcp_autolowat_cb *cb,
+ struct bpf_dynptr *dptr,
+ u32 seq, u32 end_seq)
+{
+ struct rpc_descriptor *rpc_desc;
+ u32 rpc_copied_seq;
+ u64 copy_len; /* u32 should work, but not for no_alu32 :/ */
+ u64 rpc_len;
+ int err;
+
+ rpc_copied_seq = cb->rpc_desc_seq + cb->rpc_desc_buff_len;
+
+ if (before(cb->rpc_desc_seq + RPC_DESC_SIZE, end_seq))
+ copy_len = RPC_DESC_SIZE - cb->rpc_desc_buff_len;
+ else
+ copy_len = end_seq - rpc_copied_seq;
+
+ if (copy_len == 0)
+ goto disable; /* FIN. */
+ if (copy_len > RPC_DESC_SIZE)
+ goto disable; /* always false, only for verifier. */
+ if (cb->rpc_desc_buff_len >= RPC_DESC_SIZE)
+ goto disable; /* always false, only for verifier. */
+
+ err = bpf_dynptr_read(cb->rpc_desc_buf + cb->rpc_desc_buff_len,
+ copy_len, dptr, rpc_copied_seq - seq, 0);
+ if (err)
+ goto disable;
+
+ cb->rpc_desc_buff_len += copy_len;
+
+ if (cb->rpc_desc_buff_len != RPC_DESC_SIZE) {
+ LOG("Copied %d bytes: rpc_desc_buff_len: %u", copy_len, cb->rpc_desc_buff_len);
+ goto partial;
+ }
+
+ rpc_desc = (struct rpc_descriptor *)cb->rpc_desc_buf;
+ rpc_len = RPC_DESC_SIZE + rpc_desc->header_len + rpc_desc->payload_len;
+
+ if (rpc_len > INT_MAX)
+ goto disable;
+
+ cb->rpc_end_seq = cb->rpc_desc_seq + rpc_len;
+
+ LOG("Copied full descriptor: rpc_desc_seq: %u, rpc_end_seq: %u, header_len: %u, payload_len: %u",
+ CB_SEQ(rpc_desc_seq), CB_SEQ(rpc_end_seq),
+ rpc_desc->header_len, rpc_desc->payload_len);
+
+ return 0;
+disable:
+ return -1;
+partial:
+ return 1;
+}
+
+static void tcp_set_autolowat(struct tcp_autolowat_cb *cb,
+ struct sock *sk)
+{
+ struct tcp_sock *tp = (struct tcp_sock *)sk;
+ u32 val; /* To handle wraparound. */
+
+ LOG("Setting rcvlowat: tp->copied_seq: %u, rpc_desc_seq: %u, rpc_end_seq: %u, rpc_desc_buff_len: %u",
+ TP_SEQ(copied_seq), CB_SEQ(rpc_desc_seq),
+ CB_SEQ(rpc_end_seq), cb->rpc_desc_buff_len);
+
+ if (before(tp->copied_seq, cb->rpc_desc_seq))
+ val = cb->rpc_desc_seq - tp->copied_seq;
+ else if (cb->rpc_desc_buff_len != RPC_DESC_SIZE)
+ val = RPC_DESC_SIZE;
+ else
+ val = cb->rpc_end_seq - tp->copied_seq;
+
+ if (val != tp->inet_conn.icsk_inet.sk.sk_rcvlowat) {
+ bpf_tcp_ops_set_rcvlowat(sk, val);
+
+ LOG("Set rcvlowat: expected: %u, actual: %d\n",
+ val, tp->inet_conn.icsk_inet.sk.sk_rcvlowat);
+ } else {
+ LOG("No need to set rcvlowat: %u\n", val);
+ }
+}
+
+static void tcp_disable_autolowat(struct sock *sk)
+{
+ bpf_tcp_ops_set_flags((struct tcp_sock *)sk, 0, BPF_TCP_OPS_FLAG_RCVQ);
+
+ bpf_tcp_ops_set_rcvlowat(sk, 1);
+
+ LOG("Disabled autolowat");
+}
+
+static void tcp_do_autolowat(struct tcp_autolowat_cb *cb,
+ struct sock *sk, struct sk_buff *skb)
+{
+ struct bpf_dynptr dptr;
+ struct tcp_skb_cb *tcb;
+ u32 seq, end_seq;
+ int ret = 0, i;
+
+ if (bpf_dynptr_from_skb((struct __sk_buff *)skb, 0, &dptr)) {
+ ret = -1;
+ goto update;
+ }
+
+ tcb = bpf_core_cast(skb->cb, struct tcp_skb_cb);
+ seq = tcb->seq;
+ end_seq = tcb->end_seq - !!(tcb->tcp_flags & TCPHDR_FIN);
+
+ LOG("Start parsing skb: seq: %u, end_seq: %u, len: %u, rpc_desc_seq: %u, rpc_end_seq: %u, rpc_buff_len: %u",
+ SEQ(seq), SEQ(end_seq), end_seq - seq,
+ CB_SEQ(rpc_desc_seq), CB_SEQ(rpc_end_seq), cb->rpc_desc_buff_len);
+
+ if (cb->rpc_desc_buff_len != RPC_DESC_SIZE) {
+ ret = tcp_parse_descriptor(cb, &dptr, seq, end_seq);
+ if (ret)
+ goto update;
+ }
+
+ i = 0;
+
+ while (1) {
+ if (i++ > MAX_RPC_DESC_PER_SKB) {
+ ret = -1;
+ break;
+ }
+
+ if (after(cb->rpc_end_seq, end_seq)) {
+ LOG("No more descriptor: rpc_end_seq: %u, end_seq: %u",
+ CB_SEQ(rpc_end_seq), SEQ(end_seq));
+ break;
+ }
+
+ cb->rpc_desc_seq = cb->rpc_end_seq;
+ cb->rpc_desc_buff_len = 0;
+
+ if (cb->rpc_end_seq == end_seq)
+ break;
+
+ LOG("Found next descriptor: rpc_end_seq: %u, end_seq: %u, len: %u",
+ CB_SEQ(rpc_end_seq), SEQ(end_seq), end_seq - cb->rpc_end_seq);
+
+ ret = tcp_parse_descriptor(cb, &dptr, seq, end_seq);
+ if (ret)
+ break;
+ }
+
+update:
+ if (ret >= 0)
+ tcp_set_autolowat(cb, sk);
+ else
+ tcp_disable_autolowat(sk);
+}
+
+SEC("struct_ops")
+void BPF_PROG(tcp_autolowat_enqueue_rcvq, struct sock *sk, struct sk_buff *skb)
+{
+ struct tcp_autolowat_cb *cb;
+
+ cb = bpf_sk_storage_get(&tcp_autolowat_map, sk, 0, 0);
+ if (!cb)
+ return;
+
+ tcp_do_autolowat(cb, sk, skb);
+}
+
+SEC("struct_ops")
+void BPF_PROG(tcp_autolowat_dequeue_rcvq, struct sock *sk)
+{
+ struct tcp_autolowat_cb *cb;
+
+ cb = bpf_sk_storage_get(&tcp_autolowat_map, sk, 0, 0);
+ if (!cb)
+ return;
+
+ tcp_set_autolowat(cb, sk);
+}
+
+SEC(".struct_ops.link")
+struct bpf_tcp_ops tcp_autolowat_ops = {
+ .enqueue_rcvq = (void *)tcp_autolowat_enqueue_rcvq,
+ .dequeue_rcvq = (void *)tcp_autolowat_dequeue_rcvq,
+};
+
+static int tcp_init_autolowat_cb(struct bpf_tcp_sock *btp)
+{
+ struct tcp_autolowat_cb *cb;
+ struct tcp_sock *tp;
+
+ cb = bpf_sk_storage_get(&tcp_autolowat_map, btp, 0,
+ BPF_SK_STORAGE_GET_F_CREATE);
+ if (!cb)
+ return -1;
+
+ tp = bpf_core_cast(btp, struct tcp_sock);
+
+ cb->rpc_desc_seq = tp->copied_seq;
+ cb->rpc_end_seq = tp->copied_seq;
+#ifdef DEBUG
+ cb->isn = tp->copied_seq;
+#endif
+
+ return bpf_tcp_ops_set_flags((struct tcp_sock *)btp,
+ BPF_TCP_OPS_FLAG_RCVQ, 0);
+}
+
+SEC("cgroup/setsockopt")
+int tcp_autolowat_setsockopt(struct bpf_sockopt *ctx)
+{
+ void *optval_end = ctx->optval_end;
+ int *optval = ctx->optval;
+ struct bpf_tcp_sock *btp;
+
+ if (ctx->level != SOL_BPF || ctx->optname != BPF_TCP_AUTOLOWAT)
+ goto out;
+
+ if (optval + 1 > optval_end)
+ return 0; /* -EPERM */
+
+ btp = bpf_tcp_sock(ctx->sk);
+ if (!btp)
+ goto out;
+
+ if (*optval && tcp_init_autolowat_cb(btp))
+ return 0; /* -EPERM */
+
+ /*
+ * BPF has consumed this option, don't call kernel
+ * setsockopt handler.
+ */
+ ctx->optlen = -1;
+out:
+ return 1;
+}
+
+char _license[] SEC("license") = "GPL";
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 21+ messages in thread