From: Kuniyuki Iwashima <kuniyu@google.com>
To: David Ahern <dsahern@kernel.org>,
Ido Schimmel <idosch@nvidia.com>,
"David S . Miller" <davem@davemloft.net>,
Eric Dumazet <edumazet@kernel.org>,
Jakub Kicinski <kuba@kernel.org>,
Paolo Abeni <pabeni@redhat.com>
Cc: Simon Horman <horms@kernel.org>,
Chris J Arges <carges@cloudflare.com>,
Kuniyuki Iwashima <kuniyu@google.com>,
Kuniyuki Iwashima <kuni1840@gmail.com>,
netdev@vger.kernel.org
Subject: [PATCH v1 net-next 5/5] ipv6: Batch rt6_uncached_list_flush_dev() for dying netns.
Date: Sun, 27 Sep 2026 20:23:42 +0000 [thread overview]
Message-ID: <20260927202429.2452589-6-kuniyu@google.com> (raw)
In-Reply-To: <20260927202429.2452589-1-kuniyu@google.com>
IPv6 uncached routes are linked to the global per-cpu lists,
rt6_uncached_list.
When unregistering a netdev, rt6_uncached_list_flush_dev()
iterates over the potentially long lists to find uncached
routes tied to the device and swap it with blackhole_netdev.
Since it is called for every device in dying netns under RTNL,
it adds O(N_dev x (N_cpu + N_route)) costs to netns dismantle.
Similar to IPv4, let's call it (almost) once per cleanup_net().
Chris J Arges verified the changes improve RTNL wait time
during cleanup_net() by 75x, quoted from [0]:
<...> I was able to confirm even greater reduction in contention
as measured by how much latency an unrelated process takes when
waiting for cleanup_net to complete. <...>
Some rough average latency numbers with 36 devices, 160k routes,
8 vCPUs:
- main: 137ms
<...>
- your patchset: 1.8ms
Reported-by: Chris J Arges <carges@cloudflare.com>
Closes: https://lore.kernel.org/netdev/20260917-hash-bucket-route-lists-v3-0-30493a37b6eb@cloudflare.com/
Link: https://lore.kernel.org/netdev/aq2B8PSfjn-xau4V@20HS2G4/ #[0]
Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
net/ipv6/route.c | 54 ++++++++++++++++++++++++++++++++++--------------
1 file changed, 39 insertions(+), 15 deletions(-)
diff --git a/net/ipv6/route.c b/net/ipv6/route.c
index 475ced827ec5..7747e4f20fee 100644
--- a/net/ipv6/route.c
+++ b/net/ipv6/route.c
@@ -135,6 +135,22 @@ struct uncached_list {
static DEFINE_PER_CPU_ALIGNED(struct uncached_list, rt6_uncached_list);
+static void rt6_uncached_list_replace(struct rt6_info *rt)
+{
+ struct net_device *dev = dst_dev(&rt->dst);
+ struct inet6_dev *rt_idev = rt->rt6i_idev;
+
+ if (rt_idev) {
+ rt->rt6i_idev = in6_dev_get(blackhole_netdev);
+ in6_dev_put(rt_idev);
+ }
+
+ rcu_assign_pointer(rt->dst.dev_rcu, blackhole_netdev);
+ netdev_ref_replace(dev, blackhole_netdev,
+ &rt->dst.dev_tracker,
+ GFP_ATOMIC);
+}
+
void rt6_uncached_list_add(struct rt6_info *rt)
{
struct uncached_list *ul = raw_cpu_ptr(&rt6_uncached_list);
@@ -143,7 +159,12 @@ void rt6_uncached_list_add(struct rt6_info *rt)
rt->dst.rt_uncached_list = ul;
spin_lock_bh(&ul->lock);
- list_add_tail(&rt->dst.rt_uncached, &ul->head);
+
+ if (check_net(dst_dev_net_rcu(&rt->dst)))
+ list_add_tail(&rt->dst.rt_uncached, &ul->head);
+ else
+ rt6_uncached_list_replace(rt);
+
spin_unlock_bh(&ul->lock);
}
@@ -162,6 +183,9 @@ static void rt6_uncached_list_flush_dev(struct net_device *dev)
{
int cpu;
+ if (dev && net_pre_exit_done(dev_net(dev)))
+ return;
+
for_each_possible_cpu(cpu) {
struct uncached_list *ul = per_cpu_ptr(&rt6_uncached_list, cpu);
struct rt6_info *rt, *safe;
@@ -173,23 +197,17 @@ static void rt6_uncached_list_flush_dev(struct net_device *dev)
list_for_each_entry_safe(rt, safe, &ul->head, dst.rt_uncached) {
struct inet6_dev *rt_idev = rt->rt6i_idev;
struct net_device *rt_dev = rt->dst.dev;
- bool handled = false;
- if (rt_idev && rt_idev->dev == dev) {
- rt->rt6i_idev = in6_dev_get(blackhole_netdev);
- in6_dev_put(rt_idev);
- handled = true;
+ if (dev) {
+ if (rt_dev != dev &&
+ (!rt_idev || rt_idev->dev != dev))
+ continue;
+ } else if (check_net(dev_net(rt_dev))) {
+ continue;
}
- if (rt_dev == dev) {
- rt->dst.dev = blackhole_netdev;
- netdev_ref_replace(rt_dev, blackhole_netdev,
- &rt->dst.dev_tracker,
- GFP_ATOMIC);
- handled = true;
- }
- if (handled)
- list_del_init(&rt->dst.rt_uncached);
+ rt6_uncached_list_replace(rt);
+ list_del_init(&rt->dst.rt_uncached);
}
spin_unlock_bh(&ul->lock);
}
@@ -6801,6 +6819,11 @@ static int __net_init ip6_route_net_init(struct net *net)
goto out;
}
+static void __net_exit ip6_route_net_pre_exit_batch(struct list_head *net_exit_list)
+{
+ rt6_uncached_list_flush_dev(NULL);
+}
+
static void __net_exit ip6_route_net_exit(struct net *net)
{
kfree(net->ipv6.fib6_null_entry);
@@ -6839,6 +6862,7 @@ static void __net_exit ip6_route_net_exit_late(struct net *net)
static struct pernet_operations ip6_route_net_ops = {
.init = ip6_route_net_init,
+ .pre_exit_batch = ip6_route_net_pre_exit_batch,
.exit = ip6_route_net_exit,
};
--
2.56.0.rc1.315.gc6ed9934b7-goog
next prev parent reply other threads:[~2026-09-27 20:24 UTC|newest]
Thread overview: 12+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-27 20:23 [PATCH v1 net-next 0/5] ip: Batch flushing uncached routes for dying netns Kuniyuki Iwashima
2026-09-27 20:23 ` [PATCH v1 net-next 1/5] net: Remove net->is_dying Kuniyuki Iwashima
2026-09-29 6:26 ` netdev-bot+sashiko
2026-09-27 20:23 ` [PATCH v1 net-next 2/5] net: Add ->pre_exit_batch() to struct pernet_operations Kuniyuki Iwashima
2026-09-29 6:26 ` netdev-bot+sashiko
2026-09-27 20:23 ` [PATCH v1 net-next 3/5] net: Track state in ops_undo_list() Kuniyuki Iwashima
2026-09-27 20:23 ` [PATCH v1 net-next 4/5] ipv4: Batch rt_flush_dev() for dying netns Kuniyuki Iwashima
2026-09-27 22:51 ` Eric Dumazet
2026-09-28 16:33 ` Kuniyuki Iwashima
2026-09-29 6:26 ` netdev-bot+sashiko
2026-09-27 20:23 ` Kuniyuki Iwashima [this message]
2026-09-29 6:26 ` [PATCH v1 net-next 5/5] ipv6: Batch rt6_uncached_list_flush_dev() " netdev-bot+sashiko
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260927202429.2452589-6-kuniyu@google.com \
--to=kuniyu@google.com \
--cc=carges@cloudflare.com \
--cc=davem@davemloft.net \
--cc=dsahern@kernel.org \
--cc=edumazet@kernel.org \
--cc=horms@kernel.org \
--cc=idosch@nvidia.com \
--cc=kuba@kernel.org \
--cc=kuni1840@gmail.com \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.