All of lore.kernel.org
 help / color / mirror / Atom feed
From: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>
To: Eric Dumazet <edumazet@google.com>
Cc: "David S . Miller" <davem@davemloft.net>,
	Jakub Kicinski <kuba@kernel.org>, Paolo Abeni <pabeni@redhat.com>,
	Simon Horman <horms@kernel.org>,
	Andrew Lunn <andrew+netdev@lunn.ch>,
	Ido Schimmel <idosch@nvidia.com>,
	Kuniyuki Iwashima <kuniyu@google.com>,
	Artem Lytkin <iprintercanon@gmail.com>,
	netdev@vger.kernel.org, eric.dumazet@gmail.com
Subject: Re: [PATCH net-next 5/9] sit: convert 6RD configuration to RCU protection
Date: Mon, 7 Sep 2026 15:00:20 +0200	[thread overview]
Message-ID: <ap61ZOQyfReeUxG7@lore-desk> (raw)
In-Reply-To: <20260907075846.2913645-6-edumazet@google.com>

[-- Attachment #1: Type: text/plain, Size: 12418 bytes --]

> In order to allow lockless readers in future patches, convert
> 'tunnel->ip6rd' to an RCU protected pointer.
> 
> Updating 6RD configuration via ipip6_tunnel_update_6rd() or
> ipip6_tunnel_clone_6rd() now allocates a struct ip_tunnel_6rd_parm and
> uses rcu_assign_pointer() to publish it, freeing the previous
> parameters with kfree_rcu().
> 
> Readers in check_6rd() and only_dnatted() use rcu_dereference() under
> existing RCU read lock, preventing torn reads on the 128-bit IPv6
> prefix.
> 
> Signed-off-by: Eric Dumazet <edumazet@google.com>

Hi Eric,

just few nits inline.

Acked-by: Lorenzo Bianconi <lorenzo.bianconi@oss.qualcomm.com>

Regards,
Lorenzo

> ---
>  include/net/ip_tunnels.h |   3 +-
>  net/ipv6/sit.c           | 160 ++++++++++++++++++++++++++++-----------
>  2 files changed, 117 insertions(+), 46 deletions(-)
> 
> diff --git a/include/net/ip_tunnels.h b/include/net/ip_tunnels.h
> index 7c9aadfe8fe396da10a47e93499a97141ac04f4c..7fff59bab53b682ac0d1efd1f47082ba7d633fd2 100644
> --- a/include/net/ip_tunnels.h
> +++ b/include/net/ip_tunnels.h
> @@ -127,6 +127,7 @@ struct ip_tunnel_6rd_parm {
>  	__be32			relay_prefix;
>  	u16			prefixlen;
>  	u16			relay_prefixlen;
> +	struct rcu_head		rcu;
>  };
>  #endif
>  
> @@ -185,7 +186,7 @@ struct ip_tunnel {
>  
>  	/* for SIT */
>  #ifdef CONFIG_IPV6_SIT_6RD
> -	struct ip_tunnel_6rd_parm ip6rd;
> +	struct ip_tunnel_6rd_parm __rcu *ip6rd;
>  #endif
>  	struct ip_tunnel_prl_entry __rcu *prl;	/* potential router list */
>  	unsigned int		prl_count;	/* # of entries in PRL */
> diff --git a/net/ipv6/sit.c b/net/ipv6/sit.c
> index e85fa80e80d1ad2c593b860685b9f5bac21e66d9..7cabcd3afbc4f06d9a08374e6d8df2fb1bfb9676 100644
> --- a/net/ipv6/sit.c
> +++ b/net/ipv6/sit.c
> @@ -183,21 +183,47 @@ static void ipip6_tunnel_link(struct sit_net *sitn, struct ip_tunnel *t)
>  	rcu_assign_pointer(*tp, t);
>  }
>  
> -static void ipip6_tunnel_clone_6rd(struct net_device *dev, struct sit_net *sitn)
> +static int ipip6_tunnel_clone_6rd(struct net_device *dev, struct sit_net *sitn)
>  {
>  #ifdef CONFIG_IPV6_SIT_6RD
>  	struct ip_tunnel *t = netdev_priv(dev);
> +	struct ip_tunnel_6rd_parm *new_6rd, *old_6rd;

RCT

> +
> +	new_6rd = kmalloc_obj(*new_6rd);

what about using kzalloc_obj() and get rid of the 0 initialization? This is
just control path, so I guess it is fine.

> +	if (!new_6rd)
> +		return -ENOMEM;
>  
>  	if (dev == sitn->fb_tunnel_dev || !sitn->fb_tunnel_dev) {
> -		ipv6_addr_set(&t->ip6rd.prefix, htonl(0x20020000), 0, 0, 0);
> -		t->ip6rd.relay_prefix = 0;
> -		t->ip6rd.prefixlen = 16;
> -		t->ip6rd.relay_prefixlen = 0;
> +		ipv6_addr_set(&new_6rd->prefix, htonl(0x20020000), 0, 0, 0);
> +		new_6rd->relay_prefix = 0;
> +		new_6rd->prefixlen = 16;
> +		new_6rd->relay_prefixlen = 0;
>  	} else {
>  		struct ip_tunnel *t0 = netdev_priv(sitn->fb_tunnel_dev);
> -		memcpy(&t->ip6rd, &t0->ip6rd, sizeof(t->ip6rd));
> +		struct ip_tunnel_6rd_parm *t0_6rd;
> +
> +		t0_6rd = rtnl_dereference(t0->ip6rd);
> +		if (t0_6rd) {
> +			*new_6rd = *t0_6rd;
> +		} else {
> +			ipv6_addr_set(&new_6rd->prefix, htonl(0x20020000), 0, 0, 0);
> +			new_6rd->relay_prefix = 0;
> +			new_6rd->prefixlen = 16;
> +			new_6rd->relay_prefixlen = 0;
> +		}
> +	}
> +
> +	old_6rd = rcu_dereference_protected(t->ip6rd,
> +					    lockdep_rtnl_is_held() ||
> +					    dev->reg_state == NETREG_UNINITIALIZED);
> +	rcu_assign_pointer(t->ip6rd, new_6rd);

what about using rcu_replace_pointer()?

> +	if (old_6rd) {
> +		dst_cache_reset(&t->dst_cache);
> +		netdev_state_change(t->dev);
> +		kfree_rcu(old_6rd, rcu);
>  	}
>  #endif
> +	return 0;
>  }
>  
>  static int ipip6_tunnel_create(struct net_device *dev)
> @@ -206,6 +232,10 @@ static int ipip6_tunnel_create(struct net_device *dev)
>  	struct sit_net *sitn = net_generic(t->net, sit_net_id);
>  	int err;
>  
> +	err = ipip6_tunnel_clone_6rd(dev, sitn);
> +	if (err < 0)
> +		goto out;

I guess you can just return err here.

> +
>  	__dev_addr_set(dev, &t->parms.iph.saddr, 4);
>  	memcpy(dev->broadcast, &t->parms.iph.daddr, 4);
>  
> @@ -218,8 +248,6 @@ static int ipip6_tunnel_create(struct net_device *dev)
>  	if (err < 0)
>  		goto out;
>  
> -	ipip6_tunnel_clone_6rd(dev, sitn);
> -
>  	ipip6_tunnel_link(sitn, t);
>  	return 0;
>  
> @@ -281,6 +309,7 @@ static struct ip_tunnel *ipip6_tunnel_locate(struct net *net,
>  	return nt;
>  
>  failed_free:
> +	ipip6_dev_free(dev);
>  	free_netdev(dev);
>  failed:
>  	return NULL;
> @@ -631,8 +660,13 @@ static bool only_dnatted(const struct ip_tunnel *tunnel,
>  	int prefix_len;
>  
>  #ifdef CONFIG_IPV6_SIT_6RD
> -	prefix_len = tunnel->ip6rd.prefixlen + 32
> -		- tunnel->ip6rd.relay_prefixlen;
> +	const struct ip_tunnel_6rd_parm *ip6rd;
> +
> +	ip6rd = rcu_dereference(tunnel->ip6rd);
> +	if (ip6rd)
> +		prefix_len = ip6rd->prefixlen + 32 - ip6rd->relay_prefixlen;
> +	else
> +		prefix_len = 48;
>  #else
>  	prefix_len = 48;
>  #endif
> @@ -810,25 +844,28 @@ static bool check_6rd(struct ip_tunnel *tunnel, const struct in6_addr *v6dst,
>  		      __be32 *v4dst)
>  {
>  #ifdef CONFIG_IPV6_SIT_6RD
> -	if (ipv6_prefix_equal(v6dst, &tunnel->ip6rd.prefix,
> -			      tunnel->ip6rd.prefixlen)) {
> +	const struct ip_tunnel_6rd_parm *ip6rd;
> +
> +	ip6rd = rcu_dereference(tunnel->ip6rd);
> +	if (ip6rd && ipv6_prefix_equal(v6dst, &ip6rd->prefix,
> +				       ip6rd->prefixlen)) {
>  		unsigned int pbw0, pbi0;
>  		int pbi1;
>  		u32 d;
>  
> -		pbw0 = tunnel->ip6rd.prefixlen >> 5;
> -		pbi0 = tunnel->ip6rd.prefixlen & 0x1f;
> +		pbw0 = ip6rd->prefixlen >> 5;
> +		pbi0 = ip6rd->prefixlen & 0x1f;
>  
> -		d = tunnel->ip6rd.relay_prefixlen < 32 ?
> +		d = ip6rd->relay_prefixlen < 32 ?
>  			(ntohl(v6dst->s6_addr32[pbw0]) << pbi0) >>
> -		    tunnel->ip6rd.relay_prefixlen : 0;
> +		    ip6rd->relay_prefixlen : 0;
>  
> -		pbi1 = pbi0 - tunnel->ip6rd.relay_prefixlen;
> +		pbi1 = pbi0 - ip6rd->relay_prefixlen;
>  		if (pbi1 > 0)
>  			d |= ntohl(v6dst->s6_addr32[pbw0 + 1]) >>
>  			     (32 - pbi1);
>  
> -		*v4dst = tunnel->ip6rd.relay_prefix | htonl(d);
> +		*v4dst = ip6rd->relay_prefix | htonl(d);
>  		return true;
>  	}
>  #else
> @@ -1164,6 +1201,7 @@ static void ipip6_tunnel_update(struct ip_tunnel *t,
>  static int ipip6_tunnel_update_6rd(struct ip_tunnel *t,
>  				   struct ip_tunnel_6rd *ip6rd)
>  {
> +	struct ip_tunnel_6rd_parm *new_6rd, *old_6rd;
>  	struct in6_addr prefix;
>  	__be32 relay_prefix;
>  
> @@ -1183,10 +1221,20 @@ static int ipip6_tunnel_update_6rd(struct ip_tunnel *t,
>  	if (relay_prefix != ip6rd->relay_prefix)
>  		return -EINVAL;
>  
> -	t->ip6rd.prefix = prefix;
> -	t->ip6rd.relay_prefix = relay_prefix;
> -	t->ip6rd.prefixlen = ip6rd->prefixlen;
> -	t->ip6rd.relay_prefixlen = ip6rd->relay_prefixlen;
> +	new_6rd = kmalloc_obj(*new_6rd);
> +	if (!new_6rd)
> +		return -ENOMEM;
> +
> +	new_6rd->prefix = prefix;
> +	new_6rd->relay_prefix = relay_prefix;
> +	new_6rd->prefixlen = ip6rd->prefixlen;
> +	new_6rd->relay_prefixlen = ip6rd->relay_prefixlen;
> +
> +	old_6rd = rtnl_dereference(t->ip6rd);
> +	rcu_assign_pointer(t->ip6rd, new_6rd);

rcu_replace_pointer()?

> +	if (old_6rd)
> +		kfree_rcu(old_6rd, rcu);
> +
>  	dst_cache_reset(&t->dst_cache);
>  	netdev_state_change(t->dev);
>  	return 0;
> @@ -1195,6 +1243,7 @@ static int ipip6_tunnel_update_6rd(struct ip_tunnel *t,
>  static int
>  ipip6_tunnel_get6rd(struct net_device *dev, struct ip_tunnel_parm __user *data)
>  {
> +	const struct ip_tunnel_6rd_parm *ip6rd_parm;
>  	struct ip_tunnel *t = netdev_priv(dev);
>  	struct ip_tunnel_parm_kern p;
>  	struct ip_tunnel_6rd ip6rd;
> @@ -1207,10 +1256,15 @@ ipip6_tunnel_get6rd(struct net_device *dev, struct ip_tunnel_parm __user *data)
>  	if (!t)
>  		t = netdev_priv(dev);
>  
> -	ip6rd.prefix = t->ip6rd.prefix;
> -	ip6rd.relay_prefix = t->ip6rd.relay_prefix;
> -	ip6rd.prefixlen = t->ip6rd.prefixlen;
> -	ip6rd.relay_prefixlen = t->ip6rd.relay_prefixlen;
> +	ip6rd_parm = rtnl_dereference(t->ip6rd);
> +	if (ip6rd_parm) {
> +		ip6rd.prefix = ip6rd_parm->prefix;
> +		ip6rd.relay_prefix = ip6rd_parm->relay_prefix;
> +		ip6rd.prefixlen = ip6rd_parm->prefixlen;
> +		ip6rd.relay_prefixlen = ip6rd_parm->relay_prefixlen;
> +	} else {
> +		memset(&ip6rd, 0, sizeof(ip6rd));
> +	}
>  	if (copy_to_user(data, &ip6rd, sizeof(ip6rd)))
>  		return -EFAULT;
>  	return 0;
> @@ -1222,20 +1276,16 @@ ipip6_tunnel_6rdctl(struct net_device *dev, struct ip_tunnel_6rd __user *data,
>  {
>  	struct ip_tunnel *t = netdev_priv(dev);
>  	struct ip_tunnel_6rd ip6rd;
> -	int err;
>  
>  	if (!ns_capable(t->net->user_ns, CAP_NET_ADMIN))
>  		return -EPERM;
>  	if (copy_from_user(&ip6rd, data, sizeof(ip6rd)))
>  		return -EFAULT;
>  
> -	if (cmd != SIOCDEL6RD) {
> -		err = ipip6_tunnel_update_6rd(t, &ip6rd);
> -		if (err < 0)
> -			return err;
> -	} else
> -		ipip6_tunnel_clone_6rd(dev, dev_to_sit_net(dev));
> -	return 0;
> +	if (cmd != SIOCDEL6RD)
> +		return ipip6_tunnel_update_6rd(t, &ip6rd);
> +
> +	return ipip6_tunnel_clone_6rd(dev, dev_to_sit_net(dev));
>  }
>  
>  #endif /* CONFIG_IPV6_SIT_6RD */
> @@ -1406,8 +1456,17 @@ static const struct net_device_ops ipip6_netdev_ops = {
>  static void ipip6_dev_free(struct net_device *dev)
>  {
>  	struct ip_tunnel *tunnel = netdev_priv(dev);
> +#ifdef CONFIG_IPV6_SIT_6RD
> +	struct ip_tunnel_6rd_parm *ip6rd;
>  
> -	dst_cache_destroy(&tunnel->dst_cache);
> +	ip6rd = rcu_dereference_protected(tunnel->ip6rd, 1);
> +	RCU_INIT_POINTER(tunnel->ip6rd, NULL);
> +	kfree(ip6rd);
> +#endif
> +	if (tunnel->dst_cache.cache) {
> +		dst_cache_destroy(&tunnel->dst_cache);
> +		tunnel->dst_cache.cache = NULL;
> +	}
>  }
>  
>  #define SIT_FEATURES (NETIF_F_SG	   | \
> @@ -1576,8 +1635,10 @@ static int ipip6_newlink(struct net_device *dev,
>  		return -EEXIST;
>  
>  	err = ipip6_tunnel_create(dev);
> -	if (err < 0)
> +	if (err < 0) {
> +		ipip6_dev_free(dev);
>  		return err;
> +	}
>  
>  	if (tb[IFLA_MTU]) {
>  		u32 mtu = nla_get_u32(tb[IFLA_MTU]);
> @@ -1695,6 +1756,9 @@ static int ipip6_fill_info(struct sk_buff *skb, const struct net_device *dev)
>  {
>  	struct ip_tunnel *tunnel = netdev_priv(dev);
>  	struct ip_tunnel_parm_kern *parm = &tunnel->parms;
> +#ifdef CONFIG_IPV6_SIT_6RD
> +	const struct ip_tunnel_6rd_parm *ip6rd;
> +#endif
>  
>  	if (nla_put_u32(skb, IFLA_IPTUN_LINK, parm->link) ||
>  	    nla_put_in_addr(skb, IFLA_IPTUN_LOCAL, parm->iph.saddr) ||
> @@ -1710,14 +1774,16 @@ static int ipip6_fill_info(struct sk_buff *skb, const struct net_device *dev)
>  		goto nla_put_failure;
>  
>  #ifdef CONFIG_IPV6_SIT_6RD
> -	if (nla_put_in6_addr(skb, IFLA_IPTUN_6RD_PREFIX,
> -			     &tunnel->ip6rd.prefix) ||
> -	    nla_put_in_addr(skb, IFLA_IPTUN_6RD_RELAY_PREFIX,
> -			    tunnel->ip6rd.relay_prefix) ||
> -	    nla_put_u16(skb, IFLA_IPTUN_6RD_PREFIXLEN,
> -			tunnel->ip6rd.prefixlen) ||
> -	    nla_put_u16(skb, IFLA_IPTUN_6RD_RELAY_PREFIXLEN,
> -			tunnel->ip6rd.relay_prefixlen))
> +	ip6rd = rcu_dereference_rtnl(tunnel->ip6rd);
> +	if (ip6rd &&
> +	    (nla_put_in6_addr(skb, IFLA_IPTUN_6RD_PREFIX,
> +			      &ip6rd->prefix) ||
> +	     nla_put_in_addr(skb, IFLA_IPTUN_6RD_RELAY_PREFIX,
> +			     ip6rd->relay_prefix) ||
> +	     nla_put_u16(skb, IFLA_IPTUN_6RD_PREFIXLEN,
> +			 ip6rd->prefixlen) ||
> +	     nla_put_u16(skb, IFLA_IPTUN_6RD_RELAY_PREFIXLEN,
> +			 ip6rd->relay_prefixlen)))
>  		goto nla_put_failure;
>  #endif
>  
> @@ -1863,17 +1929,21 @@ static int __net_init sit_init_net(struct net *net)
>  	t = netdev_priv(sitn->fb_tunnel_dev);
>  	t->net = net;
>  
> +	err = ipip6_tunnel_clone_6rd(sitn->fb_tunnel_dev, sitn);
> +	if (err < 0)
> +		goto err_reg_dev;
> +
>  	err = register_netdev(sitn->fb_tunnel_dev);
>  	if (err)
>  		goto err_reg_dev;
>  
> -	ipip6_tunnel_clone_6rd(sitn->fb_tunnel_dev, sitn);
>  	ipip6_fb_tunnel_init(sitn->fb_tunnel_dev);
>  
>  	strscpy(t->parms.name, sitn->fb_tunnel_dev->name);
>  	return 0;
>  
>  err_reg_dev:
> +	ipip6_dev_free(sitn->fb_tunnel_dev);
>  	free_netdev(sitn->fb_tunnel_dev);
>  err_alloc_dev:
>  	return err;
> -- 
> 2.55.0.979.g7e5102b832-goog
> 

[-- Attachment #2: signature.asc --]
[-- Type: application/pgp-signature, Size: 228 bytes --]

  reply	other threads:[~2026-09-07 13:00 UTC|newest]

Thread overview: 28+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-07  7:58 [PATCH net-next 0/9] sit: convert configuration to RCU and lockless fill_info Eric Dumazet
2026-09-07  7:58 ` [PATCH net-next 1/9] sit: fix UAF in ipip6_tunnel_del_prl() Eric Dumazet
2026-09-07 12:28   ` Lorenzo Bianconi
2026-09-07  7:58 ` [PATCH net-next 2/9] sit: charge ip_tunnel_prl_entry allocations to memcg Eric Dumazet
2026-09-07 12:35   ` Lorenzo Bianconi
2026-09-07  7:58 ` [PATCH net-next 3/9] ip_tunnel: use WRITE_ONCE in ip_tunnel_encap_setup Eric Dumazet
2026-09-07 15:12   ` Lorenzo Bianconi
2026-09-08 11:00   ` netdev-bot+sashiko
2026-09-07  7:58 ` [PATCH net-next 4/9] sit: annotate data-races around tunnel->fwmark Eric Dumazet
2026-09-07 15:12   ` Lorenzo Bianconi
2026-09-08 11:00   ` netdev-bot+sashiko
2026-09-07  7:58 ` [PATCH net-next 5/9] sit: convert 6RD configuration to RCU protection Eric Dumazet
2026-09-07 13:00   ` Lorenzo Bianconi [this message]
2026-09-08 11:00   ` netdev-bot+sashiko
2026-09-07  7:58 ` [PATCH net-next 6/9] sit: implement ipip6_get_iflink() Eric Dumazet
2026-09-07 13:01   ` Lorenzo Bianconi
2026-09-07  7:58 ` [PATCH net-next 7/9] sit: dynamically allocate struct ip_tunnel_parm_kern Eric Dumazet
2026-09-07 13:16   ` Lorenzo Bianconi
2026-09-07 13:34   ` Artem Lytkin
2026-09-07 13:48     ` Eric Dumazet
2026-09-08 11:00   ` netdev-bot+sashiko
2026-09-07  7:58 ` [PATCH net-next 8/9] sit: convert configuration to RCU protection Eric Dumazet
2026-09-07 14:32   ` Lorenzo Bianconi
2026-09-07 14:42     ` Eric Dumazet
2026-09-08 11:00   ` netdev-bot+sashiko
2026-09-07  7:58 ` [PATCH net-next 9/9] sit: no longer rely on RTNL in ipip6_fill_info() Eric Dumazet
2026-09-07 15:04   ` Lorenzo Bianconi
2026-09-11  1:40 ` [PATCH net-next 0/9] sit: convert configuration to RCU and lockless fill_info patchwork-bot+netdevbpf

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=ap61ZOQyfReeUxG7@lore-desk \
    --to=lorenzo.bianconi@oss.qualcomm.com \
    --cc=andrew+netdev@lunn.ch \
    --cc=davem@davemloft.net \
    --cc=edumazet@google.com \
    --cc=eric.dumazet@gmail.com \
    --cc=horms@kernel.org \
    --cc=idosch@nvidia.com \
    --cc=iprintercanon@gmail.com \
    --cc=kuba@kernel.org \
    --cc=kuniyu@google.com \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.