Netdev List
 help / color / mirror / Atom feed
From: Kuniyuki Iwashima <kuniyu@google.com>
To: weir@nebusec.ai
Cc: davem@davemloft.net, dsahern@kernel.org, edumazet@google.com,
	 horms@kernel.org, idosch@nvidia.com, kuba@kernel.org,
	netdev@vger.kernel.org,  pabeni@redhat.com, petalzu987@gmail.com,
	tom@herbertland.com, vega@nebusec.ai
Subject: Re: [PATCH net v2 1/1] ip6_tunnel: snapshot encap in xmit
Date: Sat,  8 Aug 2026 19:38:52 +0000	[thread overview]
Message-ID: <20260808194001.853434-1-kuniyu@google.com> (raw)
In-Reply-To: <b58876297f7d45de008f2e94b6ecab8b2ed84d21.1786088695.git.petalzu987@gmail.com>

From: Ren Wei <weir@nebusec.ai>
Date: Sat,  8 Aug 2026 16:40:49 +0800
> From: Zixuan Chai <petalzu987@gmail.com>
> 
> ip6_tnl_changelink() can update encapsulation parameters while the
> netdevice is transmitting packets. ip6_tnl_xmit() can calculate packet
> headroom with t->encap_hlen and later build an encapsulation header from
> the live t->encap. A concurrent update can change the encapsulation
> header between these accesses and make skb_push() underflow the skb head.
> 
> Take a local snapshot of t->encap before calculating the encapsulation
> header length.

This intorduce per-skb cost in the fast path for unlikely changelink.

Right approach is to convert it to RCU pointer (and remove
synchronize_net() there).

0ba269933f73 geneve: convert config to RCU-protected pointer
777434f53e77 geneve: pass geneve_config pointer to helper functions


> Use that same snapshot for headroom accounting, metadata
> validation, and build_header(). This keeps all encapsulation decisions
> for an skb consistent even if changelink updates the live configuration.
> 
> Fixes: b3a27b519b22 ("ip6_tunnel: Add support for fou/gue encapsulation")
> Cc: stable@vger.kernel.org
> Reported-by: Vega <vega@nebusec.ai>
> Assisted-by: Codex:gpt-5.4
> Signed-off-by: Zixuan Chai <petalzu987@gmail.com>
> Signed-off-by: Ren Wei <weir@nebusec.ai>
> ---
>  include/net/ip6_tunnel.h | 10 +++++-----
>  net/ipv6/ip6_tunnel.c    | 21 ++++++++++++++++-----
>  2 files changed, 21 insertions(+), 10 deletions(-)
> 
> diff --git a/include/net/ip6_tunnel.h b/include/net/ip6_tunnel.h
> index b99805ee2fd1..6e76e50a4406 100644
> --- a/include/net/ip6_tunnel.h
> +++ b/include/net/ip6_tunnel.h
> @@ -106,22 +106,22 @@ static inline int ip6_encap_hlen(struct ip_tunnel_encap *e)
>  	return hlen;
>  }
>  
> -static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip6_tnl *t,
> +static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip_tunnel_encap *e,
>  				u8 *protocol, struct flowi6 *fl6)
>  {
>  	const struct ip6_tnl_encap_ops *ops;
>  	int ret = -EINVAL;
>  
> -	if (t->encap.type == TUNNEL_ENCAP_NONE)
> +	if (e->type == TUNNEL_ENCAP_NONE)
>  		return 0;
>  
> -	if (t->encap.type >= MAX_IPTUN_ENCAP_OPS)
> +	if (e->type >= MAX_IPTUN_ENCAP_OPS)
>  		return -EINVAL;
>  
>  	rcu_read_lock();
> -	ops = rcu_dereference(ip6tun_encaps[t->encap.type]);
> +	ops = rcu_dereference(ip6tun_encaps[e->type]);
>  	if (likely(ops && ops->build_header))
> -		ret = ops->build_header(skb, &t->encap, protocol, fl6);
> +		ret = ops->build_header(skb, e, protocol, fl6);
>  	rcu_read_unlock();
>  
>  	return ret;
> diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c
> index ebf83f090376..d47757e8a388 100644
> --- a/net/ipv6/ip6_tunnel.c
> +++ b/net/ipv6/ip6_tunnel.c
> @@ -1102,6 +1102,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  		 __u8 proto)
>  {
>  	struct ip6_tnl *t = netdev_priv(dev);
> +	struct ip_tunnel_encap ipencap;
>  	struct net *net = t->net;
>  	struct ipv6hdr *ipv6h;
>  	struct ipv6_tel_txoption opt;
> @@ -1109,10 +1110,11 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	struct net_device *tdev;
>  	int err_count, mtu;
>  	unsigned int eth_hlen = t->dev->type == ARPHRD_ETHER ? ETH_HLEN : 0;
> -	unsigned int psh_hlen = sizeof(struct ipv6hdr) + t->encap_hlen;
> -	unsigned int max_headroom = psh_hlen;
> +	unsigned int max_headroom;
>  	__be16 payload_protocol;
>  	bool use_cache = false;
> +	unsigned int psh_hlen;
> +	int encap_hlen;
>  	u8 hop_limit;
>  	int err = -1;
>  
> @@ -1202,6 +1204,15 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  				     t->parms.name);
>  		goto tx_err_dst_release;
>  	}
> +
> +	/* Can tear, but hlen and build_header() use the same snapshot. */
> +	ipencap = data_race(t->encap);
> +	encap_hlen = ip6_encap_hlen(&ipencap);
> +	if (unlikely(encap_hlen < 0))
> +		goto tx_err_dst_release;
> +	psh_hlen = sizeof(struct ipv6hdr) + encap_hlen;
> +	max_headroom = psh_hlen;
> +
>  	mtu = dst6_mtu(dst) - eth_hlen - psh_hlen - t->tun_hlen;
>  	if (encap_limit >= 0) {
>  		max_headroom += 8;
> @@ -1251,7 +1262,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	}
>  
>  	if (t->parms.collect_md) {
> -		if (t->encap.type != TUNNEL_ENCAP_NONE)
> +		if (ipencap.type != TUNNEL_ENCAP_NONE)
>  			goto tx_err_dst_release;
>  	} else {
>  		if (use_cache && ndst)
> @@ -1272,10 +1283,10 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	 * needed_headroom if necessary.
>  	 */
>  	max_headroom = LL_RESERVED_SPACE(tdev) + sizeof(struct ipv6hdr)
> -			+ dst->header_len + t->hlen;
> +			+ dst->header_len + t->tun_hlen + encap_hlen;
>  	ip_tunnel_adj_headroom(dev, max_headroom);
>  
> -	err = ip6_tnl_encap(skb, t, &proto, fl6);
> +	err = ip6_tnl_encap(skb, &ipencap, &proto, fl6);
>  	if (err)
>  		return err;
>  
> -- 
> 2.34.1

  reply	other threads:[~2026-08-08 19:40 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-08  8:40 [PATCH net v2 0/1] ip6_tunnel: snapshot encap in xmit Ren Wei
2026-08-08  8:40 ` [PATCH net v2 1/1] " Ren Wei
2026-08-08 19:38   ` Kuniyuki Iwashima [this message]
2026-08-09 12:33     ` Ido Schimmel
2026-08-09 13:46   ` Ido Schimmel

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260808194001.853434-1-kuniyu@google.com \
    --to=kuniyu@google.com \
    --cc=davem@davemloft.net \
    --cc=dsahern@kernel.org \
    --cc=edumazet@google.com \
    --cc=horms@kernel.org \
    --cc=idosch@nvidia.com \
    --cc=kuba@kernel.org \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    --cc=petalzu987@gmail.com \
    --cc=tom@herbertland.com \
    --cc=vega@nebusec.ai \
    --cc=weir@nebusec.ai \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox