* [PATCH net v2 1/1] net: cap advertised device headroom
@ 2026-07-31 6:21 Zhiling Zou
2026-08-01 6:22 ` sashiko-bot
2026-08-04 11:52 ` Ido Schimmel
0 siblings, 2 replies; 3+ messages in thread
From: Zhiling Zou @ 2026-07-31 6:21 UTC (permalink / raw)
To: idosch, bpf, netdev
Cc: daniel, razor, andrew+netdev, davem, edumazet, kuba, pabeni,
dsahern, horms, yuehaibing, kuniyu, aleksander.lobakin,
maoyixie.tju, thorsten.blum, kees, vega, zhilinz
IP tunnel and netkit devices can publish headroom values that are derived
from user-created stacked devices or directly supplied netlink attributes.
These values are later used by IP output paths when reserving link-layer
space before storing skb header offsets, which are 16-bit fields.
The dynamic tunnel transmit path already caps growing needed_headroom at
512. Apply the same limit when tunnel link configuration publishes
needed_headroom or hard_header_len from an underlying route, and reject
netkit devices created with IFLA_NETKIT_HEADROOM above that value.
Keep bogus device headroom out of the control path instead of checking
every affected IP send path.
Fixes: 1a37e412a022 ("net: Use 16bits for *_headers fields of struct skbuff")
Cc: stable@vger.kernel.org
Reported-by: Vega <vega@nebusec.ai>
Signed-off-by: Zhiling Zou <zhilinz@nebusec.ai>
---
changes in v2:
- Move the fix from IP send paths to tunnel and netkit device control paths.
- Cap advertised IP tunnel headroom at 512 and reject netkit headroom
values above that limit at device creation.
- v1 Link: https://lore.kernel.org/all/0c6c64e9bbd71a0decc8504a384061e0e631be13.1785054561.git.zhilinz@nebusec.ai/
drivers/net/netkit.c | 5 ++++-
include/net/ip_tunnels.h | 11 +++++++++--
net/ipv4/ip_tunnel.c | 2 +-
net/ipv6/ip6_gre.c | 6 ++++--
net/ipv6/ip6_tunnel.c | 5 +++--
net/ipv6/sit.c | 2 +-
6 files changed, 22 insertions(+), 9 deletions(-)
diff --git a/drivers/net/netkit.c b/drivers/net/netkit.c
index a3931cd821321..d8e9ce1c659d0 100644
--- a/drivers/net/netkit.c
+++ b/drivers/net/netkit.c
@@ -25,6 +25,8 @@
#define NETKIT_NUM_RX_QUEUES_REAL 1
#define NETKIT_NUM_TX_QUEUES_REAL 1
+#define NETKIT_HEADROOM_MAX 512
+
struct netkit {
__cacheline_group_begin(netkit_fastpath);
struct net_device __rcu *peer;
@@ -1244,7 +1246,8 @@ static const struct nla_policy netkit_policy[IFLA_NETKIT_MAX + 1] = {
[IFLA_NETKIT_MODE] = NLA_POLICY_MAX(NLA_U32, NETKIT_L3),
[IFLA_NETKIT_POLICY] = { .type = NLA_U32 },
[IFLA_NETKIT_PEER_POLICY] = { .type = NLA_U32 },
- [IFLA_NETKIT_HEADROOM] = { .type = NLA_U16 },
+ [IFLA_NETKIT_HEADROOM] = NLA_POLICY_MAX(NLA_U16,
+ NETKIT_HEADROOM_MAX),
[IFLA_NETKIT_TAILROOM] = { .type = NLA_U16 },
[IFLA_NETKIT_SCRUB] = NLA_POLICY_MAX(NLA_U32, NETKIT_SCRUB_DEFAULT),
[IFLA_NETKIT_PEER_SCRUB] = NLA_POLICY_MAX(NLA_U32, NETKIT_SCRUB_DEFAULT),
diff --git a/include/net/ip_tunnels.h b/include/net/ip_tunnels.h
index d708b66e55cda..85e3455cea259 100644
--- a/include/net/ip_tunnels.h
+++ b/include/net/ip_tunnels.h
@@ -629,8 +629,7 @@ struct metadata_dst *iptunnel_metadata_reply(struct metadata_dst *md,
int skb_tunnel_check_pmtu(struct sk_buff *skb, struct dst_entry *encap_dst,
int headroom, bool reply);
-static inline void ip_tunnel_adj_headroom(struct net_device *dev,
- unsigned int headroom)
+static inline unsigned int ip_tunnel_limit_headroom(unsigned int headroom)
{
/* we must cap headroom to some upperlimit, else pskb_expand_head
* will overflow header offsets in skb_headers_offset_update().
@@ -640,6 +639,14 @@ static inline void ip_tunnel_adj_headroom(struct net_device *dev,
if (headroom > max_allowed)
headroom = max_allowed;
+ return headroom;
+}
+
+static inline void ip_tunnel_adj_headroom(struct net_device *dev,
+ unsigned int headroom)
+{
+ headroom = ip_tunnel_limit_headroom(headroom);
+
if (headroom > READ_ONCE(dev->needed_headroom))
WRITE_ONCE(dev->needed_headroom, headroom);
}
diff --git a/net/ipv4/ip_tunnel.c b/net/ipv4/ip_tunnel.c
index 9d114bd575f92..5b1f180485d42 100644
--- a/net/ipv4/ip_tunnel.c
+++ b/net/ipv4/ip_tunnel.c
@@ -317,7 +317,7 @@ static int ip_tunnel_bind_dev(struct net_device *dev)
mtu = min(tdev->mtu, IP_MAX_MTU);
}
- dev->needed_headroom = t_hlen + hlen;
+ dev->needed_headroom = ip_tunnel_limit_headroom(t_hlen + hlen);
mtu -= t_hlen + (dev->type == ARPHRD_ETHER ? dev->hard_header_len : 0);
if (mtu < IPV4_MIN_MTU)
diff --git a/net/ipv6/ip6_gre.c b/net/ipv6/ip6_gre.c
index b843116e9b703..04ac007ebcc87 100644
--- a/net/ipv6/ip6_gre.c
+++ b/net/ipv6/ip6_gre.c
@@ -1137,8 +1137,10 @@ static void ip6gre_tnl_link_config_route(struct ip6_tnl *t, int set_mtu,
return;
if (rt->dst.dev) {
- unsigned short dst_len = rt->dst.dev->hard_header_len +
- t_hlen;
+ unsigned int dst_len;
+
+ dst_len = ip_tunnel_limit_headroom(rt->dst.dev->hard_header_len +
+ t_hlen);
if (t->dev->header_ops)
dev->hard_header_len = dst_len;
diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c
index bf8e40af60b08..403995273cbae 100644
--- a/net/ipv6/ip6_tunnel.c
+++ b/net/ipv6/ip6_tunnel.c
@@ -1522,8 +1522,9 @@ static void ip6_tnl_link_config(struct ip6_tnl *t)
tdev = __dev_get_by_index(t->net, p->link);
if (tdev) {
- dev->needed_headroom = tdev->hard_header_len +
- tdev->needed_headroom + t_hlen;
+ dev->needed_headroom = ip_tunnel_limit_headroom(tdev->hard_header_len +
+ tdev->needed_headroom +
+ t_hlen);
mtu = min_t(unsigned int, tdev->mtu, IP6_MAX_MTU);
mtu = mtu - t_hlen;
diff --git a/net/ipv6/sit.c b/net/ipv6/sit.c
index a38b24fb83842..19b7fa8d1a2a0 100644
--- a/net/ipv6/sit.c
+++ b/net/ipv6/sit.c
@@ -1131,7 +1131,7 @@ static void ipip6_tunnel_bind_dev(struct net_device *dev)
WRITE_ONCE(dev->mtu, mtu);
hlen = tdev->hard_header_len + tdev->needed_headroom;
}
- dev->needed_headroom = t_hlen + hlen;
+ dev->needed_headroom = ip_tunnel_limit_headroom(t_hlen + hlen);
}
static void ipip6_tunnel_update(struct ip_tunnel *t,
--
2.43.0
^ permalink raw reply related [flat|nested] 3+ messages in thread
* Re: [PATCH net v2 1/1] net: cap advertised device headroom
2026-07-31 6:21 [PATCH net v2 1/1] net: cap advertised device headroom Zhiling Zou
@ 2026-08-01 6:22 ` sashiko-bot
2026-08-04 11:52 ` Ido Schimmel
1 sibling, 0 replies; 3+ messages in thread
From: sashiko-bot @ 2026-08-01 6:22 UTC (permalink / raw)
To: Zhiling Zou; +Cc: bpf
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
Pre-existing issues:
- [High] The patch attempts to prevent 16-bit skb header offset overflows by capping `needed_headroom` to 512 at device configuration. However, this only mitigates overflows caused by a single massive headroom allocation. It fails to prevent cumulative overflows in the data path. An attacker can chain multiple tunnels and interpose a qdisc like `netem` to bypass the `dev_xmit_recursion()` limit (since `netem` defers transmission to a timer, unwinding the call stack and resetting the recursion counter). Because `needed_headroom` is capped, `skb_cow_head()` will repeatedly calculate a small `delta` and trigger `pskb_expand_head()`. Each call increments the 16-bit header offsets. After enough chained transmissions (e.g., ~1600 tunnels), the offsets will exceed 65535 and wrap around, leading to out-of-bounds reads or writes when the headers are accessed.
--
--- Patch [1]: [PATCH net v2 1/1] net: cap advertised device headroom ---
commit 9a8b47db3e094ff90f0e4cb79a3ba54d0ddcaab8
Author: Zhiling Zou <zhilinz@nebusec.ai>
net: cap advertised device headroom
IP tunnel and netkit devices can publish headroom values that are derived
from user-created stacked devices or directly supplied netlink attributes.
The dynamic tunnel transmit path already caps growing needed_headroom at
512.
Link: https://lore.kernel.org/all/0c6c64e9bbd71a0decc8504a384061e0e631be13.1785054561.git.zhilinz@nebusec.ai/
> diff --git a/include/net/ip_tunnels.h b/include/net/ip_tunnels.h
> index d708b66e55cda..85e3455cea259 100644
> --- a/include/net/ip_tunnels.h
> +++ b/include/net/ip_tunnels.h
> @@ -629,8 +629,7 @@ struct metadata_dst *iptunnel_metadata_reply(struct metadata_dst *md,
> int skb_tunnel_check_pmtu(struct sk_buff *skb, struct dst_entry *encap_dst,
> int headroom, bool reply);
>
> -static inline void ip_tunnel_adj_headroom(struct net_device *dev,
> - unsigned int headroom)
> +static inline unsigned int ip_tunnel_limit_headroom(unsigned int headroom)
> {
> /* we must cap headroom to some upperlimit, else pskb_expand_head
> * will overflow header offsets in skb_headers_offset_update().
> @@ -640,6 +639,14 @@ static inline void ip_tunnel_adj_headroom(struct net_device *dev,
> if (headroom > max_allowed)
> headroom = max_allowed;
>
> + return headroom;
> +}
[Severity: High]
This is a pre-existing issue, but does capping needed_headroom to 512 here
completely mitigate the 16-bit header offset overflow in
skb_headers_offset_update()?
While this prevents a single massive allocation from overflowing the offsets,
it seems it doesn't prevent cumulative overflows in the data path. If a deep
chain of network tunnels is created and a delaying qdisc (like netem) is
attached to each to bypass the dev_xmit_recursion() limit (since netem
defers transmission to a timer, unwinding the call stack and resetting
the recursion counter), skb_cow_head() could repeatedly calculate a small
delta and trigger pskb_expand_head().
Each call increments the 16-bit header offsets in
skb_headers_offset_update():
net/core/skbuff.c:skb_headers_offset_update() {
...
skb->mac_header += off;
...
}
After enough chained transmissions, could the offsets exceed 65535 and wrap
around, leading to out-of-bounds reads or writes when the network layers
access the wrapped header pointers?
--
Sashiko AI review · https://sashiko.dev/#/patchset/0ae4aa29223b89049727aec4d36f144bad41537e.1785476387.git.zhilinz@nebusec.ai?part=1
^ permalink raw reply [flat|nested] 3+ messages in thread
* Re: [PATCH net v2 1/1] net: cap advertised device headroom
2026-07-31 6:21 [PATCH net v2 1/1] net: cap advertised device headroom Zhiling Zou
2026-08-01 6:22 ` sashiko-bot
@ 2026-08-04 11:52 ` Ido Schimmel
1 sibling, 0 replies; 3+ messages in thread
From: Ido Schimmel @ 2026-08-04 11:52 UTC (permalink / raw)
To: Zhiling Zou
Cc: bpf, netdev, daniel, razor, andrew+netdev, davem, edumazet, kuba,
pabeni, dsahern, horms, yuehaibing, kuniyu, aleksander.lobakin,
maoyixie.tju, thorsten.blum, kees, vega
On Fri, Jul 31, 2026 at 02:21:43PM +0800, Zhiling Zou wrote:
> IP tunnel and netkit devices can publish headroom values that are derived
> from user-created stacked devices or directly supplied netlink attributes.
> These values are later used by IP output paths when reserving link-layer
> space before storing skb header offsets, which are 16-bit fields.
>
> The dynamic tunnel transmit path already caps growing needed_headroom at
> 512. Apply the same limit when tunnel link configuration publishes
> needed_headroom or hard_header_len from an underlying route, and reject
> netkit devices created with IFLA_NETKIT_HEADROOM above that value.
The netkit hunk should be a separate patch (it cannot blame 1a37e412a022
which came before it). The netkit maintainers might ask for a higher
ceiling than 512 bytes given that configurations that passed before will
be rejected now. Let's fix IP tunnels first and netkit afterwards.
>
> Keep bogus device headroom out of the control path instead of checking
> every affected IP send path.
The commit message should explain why capping the headroom is safe. IP
tunnels still expand the headroom if it's not enough, so the cost is
only reallocation for nonsense configurations.
>
> Fixes: 1a37e412a022 ("net: Use 16bits for *_headers fields of struct skbuff")
> Cc: stable@vger.kernel.org
> Reported-by: Vega <vega@nebusec.ai>
> Signed-off-by: Zhiling Zou <zhilinz@nebusec.ai>
> ---
> changes in v2:
> - Move the fix from IP send paths to tunnel and netkit device control paths.
> - Cap advertised IP tunnel headroom at 512 and reject netkit headroom
> values above that limit at device creation.
> - v1 Link: https://lore.kernel.org/all/0c6c64e9bbd71a0decc8504a384061e0e631be13.1785054561.git.zhilinz@nebusec.ai/
>
> drivers/net/netkit.c | 5 ++++-
> include/net/ip_tunnels.h | 11 +++++++++--
> net/ipv4/ip_tunnel.c | 2 +-
> net/ipv6/ip6_gre.c | 6 ++++--
> net/ipv6/ip6_tunnel.c | 5 +++--
> net/ipv6/sit.c | 2 +-
> 6 files changed, 22 insertions(+), 9 deletions(-)
>
> diff --git a/drivers/net/netkit.c b/drivers/net/netkit.c
> index a3931cd821321..d8e9ce1c659d0 100644
> --- a/drivers/net/netkit.c
> +++ b/drivers/net/netkit.c
> @@ -25,6 +25,8 @@
> #define NETKIT_NUM_RX_QUEUES_REAL 1
> #define NETKIT_NUM_TX_QUEUES_REAL 1
>
> +#define NETKIT_HEADROOM_MAX 512
> +
> struct netkit {
> __cacheline_group_begin(netkit_fastpath);
> struct net_device __rcu *peer;
> @@ -1244,7 +1246,8 @@ static const struct nla_policy netkit_policy[IFLA_NETKIT_MAX + 1] = {
> [IFLA_NETKIT_MODE] = NLA_POLICY_MAX(NLA_U32, NETKIT_L3),
> [IFLA_NETKIT_POLICY] = { .type = NLA_U32 },
> [IFLA_NETKIT_PEER_POLICY] = { .type = NLA_U32 },
> - [IFLA_NETKIT_HEADROOM] = { .type = NLA_U16 },
> + [IFLA_NETKIT_HEADROOM] = NLA_POLICY_MAX(NLA_U16,
> + NETKIT_HEADROOM_MAX),
> [IFLA_NETKIT_TAILROOM] = { .type = NLA_U16 },
> [IFLA_NETKIT_SCRUB] = NLA_POLICY_MAX(NLA_U32, NETKIT_SCRUB_DEFAULT),
> [IFLA_NETKIT_PEER_SCRUB] = NLA_POLICY_MAX(NLA_U32, NETKIT_SCRUB_DEFAULT),
> diff --git a/include/net/ip_tunnels.h b/include/net/ip_tunnels.h
> index d708b66e55cda..85e3455cea259 100644
> --- a/include/net/ip_tunnels.h
> +++ b/include/net/ip_tunnels.h
> @@ -629,8 +629,7 @@ struct metadata_dst *iptunnel_metadata_reply(struct metadata_dst *md,
> int skb_tunnel_check_pmtu(struct sk_buff *skb, struct dst_entry *encap_dst,
> int headroom, bool reply);
>
> -static inline void ip_tunnel_adj_headroom(struct net_device *dev,
> - unsigned int headroom)
> +static inline unsigned int ip_tunnel_limit_headroom(unsigned int headroom)
> {
> /* we must cap headroom to some upperlimit, else pskb_expand_head
> * will overflow header offsets in skb_headers_offset_update().
> @@ -640,6 +639,14 @@ static inline void ip_tunnel_adj_headroom(struct net_device *dev,
> if (headroom > max_allowed)
> headroom = max_allowed;
>
> + return headroom;
> +}
> +
> +static inline void ip_tunnel_adj_headroom(struct net_device *dev,
> + unsigned int headroom)
> +{
> + headroom = ip_tunnel_limit_headroom(headroom);
> +
> if (headroom > READ_ONCE(dev->needed_headroom))
> WRITE_ONCE(dev->needed_headroom, headroom);
> }
> diff --git a/net/ipv4/ip_tunnel.c b/net/ipv4/ip_tunnel.c
> index 9d114bd575f92..5b1f180485d42 100644
> --- a/net/ipv4/ip_tunnel.c
> +++ b/net/ipv4/ip_tunnel.c
> @@ -317,7 +317,7 @@ static int ip_tunnel_bind_dev(struct net_device *dev)
> mtu = min(tdev->mtu, IP_MAX_MTU);
> }
>
> - dev->needed_headroom = t_hlen + hlen;
> + dev->needed_headroom = ip_tunnel_limit_headroom(t_hlen + hlen);
> mtu -= t_hlen + (dev->type == ARPHRD_ETHER ? dev->hard_header_len : 0);
>
> if (mtu < IPV4_MIN_MTU)
> diff --git a/net/ipv6/ip6_gre.c b/net/ipv6/ip6_gre.c
> index b843116e9b703..04ac007ebcc87 100644
> --- a/net/ipv6/ip6_gre.c
> +++ b/net/ipv6/ip6_gre.c
> @@ -1137,8 +1137,10 @@ static void ip6gre_tnl_link_config_route(struct ip6_tnl *t, int set_mtu,
> return;
>
> if (rt->dst.dev) {
> - unsigned short dst_len = rt->dst.dev->hard_header_len +
> - t_hlen;
> + unsigned int dst_len;
> +
> + dst_len = ip_tunnel_limit_headroom(rt->dst.dev->hard_header_len +
> + t_hlen);
You can use a local variable to avoid crossing the 80 chars limit.
>
> if (t->dev->header_ops)
> dev->hard_header_len = dst_len;
> diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c
> index bf8e40af60b08..403995273cbae 100644
> --- a/net/ipv6/ip6_tunnel.c
> +++ b/net/ipv6/ip6_tunnel.c
> @@ -1522,8 +1522,9 @@ static void ip6_tnl_link_config(struct ip6_tnl *t)
> tdev = __dev_get_by_index(t->net, p->link);
>
> if (tdev) {
> - dev->needed_headroom = tdev->hard_header_len +
> - tdev->needed_headroom + t_hlen;
> + dev->needed_headroom = ip_tunnel_limit_headroom(tdev->hard_header_len +
> + tdev->needed_headroom +
> + t_hlen);
Likewise.
> mtu = min_t(unsigned int, tdev->mtu, IP6_MAX_MTU);
>
> mtu = mtu - t_hlen;
> diff --git a/net/ipv6/sit.c b/net/ipv6/sit.c
> index a38b24fb83842..19b7fa8d1a2a0 100644
> --- a/net/ipv6/sit.c
> +++ b/net/ipv6/sit.c
> @@ -1131,7 +1131,7 @@ static void ipip6_tunnel_bind_dev(struct net_device *dev)
> WRITE_ONCE(dev->mtu, mtu);
> hlen = tdev->hard_header_len + tdev->needed_headroom;
> }
> - dev->needed_headroom = t_hlen + hlen;
> + dev->needed_headroom = ip_tunnel_limit_headroom(t_hlen + hlen);
> }
>
> static void ipip6_tunnel_update(struct ip_tunnel *t,
> --
> 2.43.0
^ permalink raw reply [flat|nested] 3+ messages in thread
end of thread, other threads:[~2026-08-04 11:52 UTC | newest]
Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-07-31 6:21 [PATCH net v2 1/1] net: cap advertised device headroom Zhiling Zou
2026-08-01 6:22 ` sashiko-bot
2026-08-04 11:52 ` Ido Schimmel
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.