From: ThisSeanZhang <thisseanzhang@gmail.com>
To: bpf@vger.kernel.org
Cc: Alexei Starovoitov <ast@kernel.org>,
Daniel Borkmann <daniel@iogearbox.net>,
Andrii Nakryiko <andrii@kernel.org>,
Eduard Zingerman <eddyz87@gmail.com>,
netdev@vger.kernel.org, Nick Hudson <nhudson@akamai.com>,
Felix Fietkau <nbd@nbd.name>, Qingfang Deng <dqfext@gmail.com>
Subject: [RFC bpf-next v2 2/3] bpf: Add PPPoE decap support to bpf_skb_adjust_room
Date: Sat, 3 Oct 2026 15:33:46 -0400 [thread overview]
Message-ID: <20261003193347.1137527-3-thisseanzhang@gmail.com> (raw)
In-Reply-To: <20261003193347.1137527-1-thisseanzhang@gmail.com>
Add a new BPF_F_ADJ_ROOM_DECAP_PPPOE flag to bpf_skb_adjust_room() so
that a TC BPF program can decapsulate a PPPoE session packet:
bpf_skb_adjust_room(skb, -PPPOE_SES_HLEN, BPF_ADJ_ROOM_MAC,
BPF_F_ADJ_ROOM_DECAP_PPPOE);
The flag removes the six-byte PPPoE session header and the following
two-byte PPP protocol field between the MAC and network headers. It
restores consistent skb metadata for later consumers: the PPP protocol
field selects ETH_P_IP or ETH_P_IPV6, and any other PPP protocol is
rejected.
Before this patch, bpf_skb_adjust_room() rejected PPPoE packets because
it only accepted ETH_P_IP or ETH_P_IPV6 in skb->protocol. A TC program
therefore could not remove the PPPoE header and restore the inner
protocol metadata with the existing helper.
This is the counterpart to BPF_F_ADJ_ROOM_ENCAP_PPPOE introduced in
the previous patch and restores plain IPv4 or IPv6 skb metadata after
PPPoE decapsulation.
The flag also resets mac_len after removing the header. On flows where
a packet encapsulated on the same host re-enters TC ingress without
passing through the receive path, mac_len may still include the
removed PPPoE header. The reset mirrors what pppoe_gso_segment() does
for its segments.
Signed-off-by: ThisSeanZhang <thisseanzhang@gmail.com>
---
include/uapi/linux/bpf.h | 10 +++++
net/core/filter.c | 69 +++++++++++++++++++++++++++++++---
tools/include/uapi/linux/bpf.h | 10 +++++
3 files changed, 83 insertions(+), 6 deletions(-)
diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
index 30481d040..52a509060 100644
--- a/include/uapi/linux/bpf.h
+++ b/include/uapi/linux/bpf.h
@@ -3071,6 +3071,15 @@ union bpf_attr {
* decapsulating a tunnel with an outer IPv6 header (IPv6-in-IPv6
* or IPv4-in-IPv6).
*
+ * * **BPF_F_ADJ_ROOM_DECAP_PPPOE**:
+ * Decapsulate a PPPoE session header. Must be used with
+ * **BPF_ADJ_ROOM_MAC** mode and a negative *len_diff* equal to
+ * the size of the PPPoE session header plus the PPP protocol
+ * field (8 bytes in total). The PPP protocol field determines
+ * the new *skb->protocol* (**ETH_P_IP** or **ETH_P_IPV6**);
+ * a payload with any other PPP protocol is rejected. The Ethernet
+ * type in the MAC header is left for the BPF program to restore.
+ *
* When using the decapsulation flags above, the skb->encapsulation
* flag is automatically cleared if all tunnel-specific GSO flags
* (SKB_GSO_UDP_TUNNEL, SKB_GSO_UDP_TUNNEL_CSUM, SKB_GSO_GRE,
@@ -6335,6 +6344,7 @@ enum bpf_adj_room_flags {
BPF_F_ADJ_ROOM_DECAP_IPXIP4 = (1ULL << 11),
BPF_F_ADJ_ROOM_DECAP_IPXIP6 = (1ULL << 12),
BPF_F_ADJ_ROOM_ENCAP_PPPOE = (1ULL << 13),
+ BPF_F_ADJ_ROOM_DECAP_PPPOE = (1ULL << 14),
};
enum {
diff --git a/net/core/filter.c b/net/core/filter.c
index 5b204e316..46986a84e 100644
--- a/net/core/filter.c
+++ b/net/core/filter.c
@@ -48,6 +48,7 @@
#include <linux/seccomp.h>
#include <linux/if_vlan.h>
#include <linux/if_pppox.h>
+#include <linux/ppp_defs.h>
#include <linux/bpf.h>
#include <linux/btf.h>
#include <net/sch_generic.h>
@@ -3590,7 +3591,8 @@ static u32 bpf_skb_net_base_len(const struct sk_buff *skb)
#define BPF_F_ADJ_ROOM_DECAP_MASK (BPF_F_ADJ_ROOM_DECAP_L3_MASK | \
BPF_F_ADJ_ROOM_DECAP_L4_MASK | \
- BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK)
+ BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK | \
+ BPF_F_ADJ_ROOM_DECAP_PPPOE)
#define BPF_F_ADJ_ROOM_MASK (BPF_F_ADJ_ROOM_FIXED_GSO | \
BPF_F_ADJ_ROOM_ENCAP_MASK | \
@@ -3724,6 +3726,8 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
u64 flags)
{
bool decap = flags & BPF_F_ADJ_ROOM_DECAP_L3_MASK;
+ __be16 inner_proto = 0;
+ u32 inner_len = 0;
int ret;
if (unlikely(flags & ~(BPF_F_ADJ_ROOM_DECAP_MASK |
@@ -3742,6 +3746,33 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
if (unlikely(ret < 0))
return ret;
+ if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+ u16 ppp_proto;
+
+ if (unlikely(!pskb_may_pull(skb, off + PPPOE_SES_HLEN)))
+ return -ENOMEM;
+
+ /* PPP protocol field follows the PPPoE session header. */
+ ppp_proto = get_unaligned_be16(skb->data + off +
+ sizeof(struct pppoe_hdr));
+ switch (ppp_proto) {
+ case PPP_IP:
+ inner_proto = htons(ETH_P_IP);
+ inner_len = sizeof(struct iphdr);
+ break;
+ case PPP_IPV6:
+ inner_proto = htons(ETH_P_IPV6);
+ inner_len = sizeof(struct ipv6hdr);
+ break;
+ default:
+ return -ENOTSUPP;
+ }
+
+ /* A full inner L3 header must remain after decapsulation. */
+ if (skb->len - off - PPPOE_SES_HLEN < inner_len)
+ return -EINVAL;
+ }
+
ret = bpf_skb_net_hdr_pop(skb, off, len_diff);
if (unlikely(ret < 0))
return ret;
@@ -3757,6 +3788,13 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
skb_dst_drop(skb);
}
+ if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+ skb->protocol = inner_proto;
+ skb_reset_mac_len(skb);
+ if (skb_valid_dst(skb))
+ skb_dst_drop(skb);
+ }
+
if (skb_is_gso(skb)) {
struct skb_shared_info *shinfo = skb_shinfo(skb);
@@ -3869,9 +3907,13 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
return -EINVAL;
if (unlikely(len_diff_abs > 0xfffU))
return -EFAULT;
- if (unlikely(proto != htons(ETH_P_IP) &&
- proto != htons(ETH_P_IPV6)))
+ if (unlikely(flags & BPF_F_ADJ_ROOM_DECAP_PPPOE)) {
+ if (proto != htons(ETH_P_PPP_SES))
+ return -ENOTSUPP;
+ } else if (unlikely(proto != htons(ETH_P_IP) &&
+ proto != htons(ETH_P_IPV6))) {
return -ENOTSUPP;
+ }
off = skb_mac_header_len(skb);
switch (mode) {
@@ -3885,9 +3927,7 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
}
if (flags & BPF_F_ADJ_ROOM_ENCAP_PPPOE) {
- /* The PPPoE session header has a fixed size and is
- * inserted directly after the MAC header.
- */
+ /* Fixed-size header, inserted after the MAC header. */
if (shrink || mode != BPF_ADJ_ROOM_MAC ||
len_diff != PPPOE_SES_HLEN ||
flags & ((BPF_F_ADJ_ROOM_ENCAP_MASK |
@@ -3920,6 +3960,23 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
(flags & BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK))
return -EINVAL;
+ /* PPPoE decapsulation is mutually exclusive with the
+ * other decapsulation types.
+ */
+ if ((flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) &&
+ (flags & (BPF_F_ADJ_ROOM_DECAP_MASK &
+ ~BPF_F_ADJ_ROOM_DECAP_PPPOE)))
+ return -EINVAL;
+
+ if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+ /* Fixed-size header; require a full inner L3 header. */
+ if (mode != BPF_ADJ_ROOM_MAC ||
+ len_diff_abs != PPPOE_SES_HLEN)
+ return -EINVAL;
+
+ len_min = sizeof(struct iphdr);
+ }
+
if (flags & BPF_F_ADJ_ROOM_DECAP_L4_MASK)
len_decap_min += bpf_skb_net_base_len(skb);
diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h
index 30481d040..52a509060 100644
--- a/tools/include/uapi/linux/bpf.h
+++ b/tools/include/uapi/linux/bpf.h
@@ -3071,6 +3071,15 @@ union bpf_attr {
* decapsulating a tunnel with an outer IPv6 header (IPv6-in-IPv6
* or IPv4-in-IPv6).
*
+ * * **BPF_F_ADJ_ROOM_DECAP_PPPOE**:
+ * Decapsulate a PPPoE session header. Must be used with
+ * **BPF_ADJ_ROOM_MAC** mode and a negative *len_diff* equal to
+ * the size of the PPPoE session header plus the PPP protocol
+ * field (8 bytes in total). The PPP protocol field determines
+ * the new *skb->protocol* (**ETH_P_IP** or **ETH_P_IPV6**);
+ * a payload with any other PPP protocol is rejected. The Ethernet
+ * type in the MAC header is left for the BPF program to restore.
+ *
* When using the decapsulation flags above, the skb->encapsulation
* flag is automatically cleared if all tunnel-specific GSO flags
* (SKB_GSO_UDP_TUNNEL, SKB_GSO_UDP_TUNNEL_CSUM, SKB_GSO_GRE,
@@ -6335,6 +6344,7 @@ enum bpf_adj_room_flags {
BPF_F_ADJ_ROOM_DECAP_IPXIP4 = (1ULL << 11),
BPF_F_ADJ_ROOM_DECAP_IPXIP6 = (1ULL << 12),
BPF_F_ADJ_ROOM_ENCAP_PPPOE = (1ULL << 13),
+ BPF_F_ADJ_ROOM_DECAP_PPPOE = (1ULL << 14),
};
enum {
--
2.47.3
next prev parent reply other threads:[~2026-10-03 19:34 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-03 19:33 [RFC bpf-next v2 0/3] bpf: Add PPPoE encap/decap support to bpf_skb_adjust_room ThisSeanZhang
2026-10-03 19:33 ` [RFC bpf-next v2 1/3] bpf: Add PPPoE encap " ThisSeanZhang
2026-10-03 19:33 ` ThisSeanZhang [this message]
2026-10-03 19:33 ` [RFC bpf-next v2 3/3] selftests/bpf: Add a test for the PPPoE encap/decap adjust_room flags ThisSeanZhang
2026-10-03 20:11 ` bot+bpf-ci
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261003193347.1137527-3-thisseanzhang@gmail.com \
--to=thisseanzhang@gmail.com \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=daniel@iogearbox.net \
--cc=dqfext@gmail.com \
--cc=eddyz87@gmail.com \
--cc=nbd@nbd.name \
--cc=netdev@vger.kernel.org \
--cc=nhudson@akamai.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox