From: Kuniyuki Iwashima <kuniyu@google.com>
To: weir@nebusec.ai
Cc: davem@davemloft.net, dsahern@kernel.org, edumazet@google.com,
horms@kernel.org, idosch@nvidia.com, kuba@kernel.org,
netdev@vger.kernel.org, pabeni@redhat.com, petalzu987@gmail.com,
tom@herbertland.com, vega@nebusec.ai
Subject: Re: [PATCH net v2 1/1] ip6_tunnel: snapshot encap in xmit
Date: Sat, 8 Aug 2026 19:38:52 +0000 [thread overview]
Message-ID: <20260808194001.853434-1-kuniyu@google.com> (raw)
In-Reply-To: <b58876297f7d45de008f2e94b6ecab8b2ed84d21.1786088695.git.petalzu987@gmail.com>
From: Ren Wei <weir@nebusec.ai>
Date: Sat, 8 Aug 2026 16:40:49 +0800
> From: Zixuan Chai <petalzu987@gmail.com>
>
> ip6_tnl_changelink() can update encapsulation parameters while the
> netdevice is transmitting packets. ip6_tnl_xmit() can calculate packet
> headroom with t->encap_hlen and later build an encapsulation header from
> the live t->encap. A concurrent update can change the encapsulation
> header between these accesses and make skb_push() underflow the skb head.
>
> Take a local snapshot of t->encap before calculating the encapsulation
> header length.
This intorduce per-skb cost in the fast path for unlikely changelink.
Right approach is to convert it to RCU pointer (and remove
synchronize_net() there).
0ba269933f73 geneve: convert config to RCU-protected pointer
777434f53e77 geneve: pass geneve_config pointer to helper functions
> Use that same snapshot for headroom accounting, metadata
> validation, and build_header(). This keeps all encapsulation decisions
> for an skb consistent even if changelink updates the live configuration.
>
> Fixes: b3a27b519b22 ("ip6_tunnel: Add support for fou/gue encapsulation")
> Cc: stable@vger.kernel.org
> Reported-by: Vega <vega@nebusec.ai>
> Assisted-by: Codex:gpt-5.4
> Signed-off-by: Zixuan Chai <petalzu987@gmail.com>
> Signed-off-by: Ren Wei <weir@nebusec.ai>
> ---
> include/net/ip6_tunnel.h | 10 +++++-----
> net/ipv6/ip6_tunnel.c | 21 ++++++++++++++++-----
> 2 files changed, 21 insertions(+), 10 deletions(-)
>
> diff --git a/include/net/ip6_tunnel.h b/include/net/ip6_tunnel.h
> index b99805ee2fd1..6e76e50a4406 100644
> --- a/include/net/ip6_tunnel.h
> +++ b/include/net/ip6_tunnel.h
> @@ -106,22 +106,22 @@ static inline int ip6_encap_hlen(struct ip_tunnel_encap *e)
> return hlen;
> }
>
> -static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip6_tnl *t,
> +static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip_tunnel_encap *e,
> u8 *protocol, struct flowi6 *fl6)
> {
> const struct ip6_tnl_encap_ops *ops;
> int ret = -EINVAL;
>
> - if (t->encap.type == TUNNEL_ENCAP_NONE)
> + if (e->type == TUNNEL_ENCAP_NONE)
> return 0;
>
> - if (t->encap.type >= MAX_IPTUN_ENCAP_OPS)
> + if (e->type >= MAX_IPTUN_ENCAP_OPS)
> return -EINVAL;
>
> rcu_read_lock();
> - ops = rcu_dereference(ip6tun_encaps[t->encap.type]);
> + ops = rcu_dereference(ip6tun_encaps[e->type]);
> if (likely(ops && ops->build_header))
> - ret = ops->build_header(skb, &t->encap, protocol, fl6);
> + ret = ops->build_header(skb, e, protocol, fl6);
> rcu_read_unlock();
>
> return ret;
> diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c
> index ebf83f090376..d47757e8a388 100644
> --- a/net/ipv6/ip6_tunnel.c
> +++ b/net/ipv6/ip6_tunnel.c
> @@ -1102,6 +1102,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
> __u8 proto)
> {
> struct ip6_tnl *t = netdev_priv(dev);
> + struct ip_tunnel_encap ipencap;
> struct net *net = t->net;
> struct ipv6hdr *ipv6h;
> struct ipv6_tel_txoption opt;
> @@ -1109,10 +1110,11 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
> struct net_device *tdev;
> int err_count, mtu;
> unsigned int eth_hlen = t->dev->type == ARPHRD_ETHER ? ETH_HLEN : 0;
> - unsigned int psh_hlen = sizeof(struct ipv6hdr) + t->encap_hlen;
> - unsigned int max_headroom = psh_hlen;
> + unsigned int max_headroom;
> __be16 payload_protocol;
> bool use_cache = false;
> + unsigned int psh_hlen;
> + int encap_hlen;
> u8 hop_limit;
> int err = -1;
>
> @@ -1202,6 +1204,15 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
> t->parms.name);
> goto tx_err_dst_release;
> }
> +
> + /* Can tear, but hlen and build_header() use the same snapshot. */
> + ipencap = data_race(t->encap);
> + encap_hlen = ip6_encap_hlen(&ipencap);
> + if (unlikely(encap_hlen < 0))
> + goto tx_err_dst_release;
> + psh_hlen = sizeof(struct ipv6hdr) + encap_hlen;
> + max_headroom = psh_hlen;
> +
> mtu = dst6_mtu(dst) - eth_hlen - psh_hlen - t->tun_hlen;
> if (encap_limit >= 0) {
> max_headroom += 8;
> @@ -1251,7 +1262,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
> }
>
> if (t->parms.collect_md) {
> - if (t->encap.type != TUNNEL_ENCAP_NONE)
> + if (ipencap.type != TUNNEL_ENCAP_NONE)
> goto tx_err_dst_release;
> } else {
> if (use_cache && ndst)
> @@ -1272,10 +1283,10 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
> * needed_headroom if necessary.
> */
> max_headroom = LL_RESERVED_SPACE(tdev) + sizeof(struct ipv6hdr)
> - + dst->header_len + t->hlen;
> + + dst->header_len + t->tun_hlen + encap_hlen;
> ip_tunnel_adj_headroom(dev, max_headroom);
>
> - err = ip6_tnl_encap(skb, t, &proto, fl6);
> + err = ip6_tnl_encap(skb, &ipencap, &proto, fl6);
> if (err)
> return err;
>
> --
> 2.34.1
next prev parent reply other threads:[~2026-08-08 19:40 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-08 8:40 [PATCH net v2 0/1] ip6_tunnel: snapshot encap in xmit Ren Wei
2026-08-08 8:40 ` [PATCH net v2 1/1] " Ren Wei
2026-08-08 19:38 ` Kuniyuki Iwashima [this message]
2026-08-09 12:33 ` Ido Schimmel
2026-08-09 13:46 ` Ido Schimmel
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260808194001.853434-1-kuniyu@google.com \
--to=kuniyu@google.com \
--cc=davem@davemloft.net \
--cc=dsahern@kernel.org \
--cc=edumazet@google.com \
--cc=horms@kernel.org \
--cc=idosch@nvidia.com \
--cc=kuba@kernel.org \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
--cc=petalzu987@gmail.com \
--cc=tom@herbertland.com \
--cc=vega@nebusec.ai \
--cc=weir@nebusec.ai \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox