From: Chris J Arges <carges@cloudflare.com>
To: David Ahern <dsahern@kernel.org>,
Ido Schimmel <idosch@nvidia.com>,
"David S. Miller" <davem@davemloft.net>,
Eric Dumazet <edumazet@google.com>,
Jakub Kicinski <kuba@kernel.org>,
Paolo Abeni <pabeni@redhat.com>, Simon Horman <horms@kernel.org>,
Shuah Khan <shuah@kernel.org>
Cc: netdev@vger.kernel.org, linux-kernel@vger.kernel.org,
linux-kselftest@vger.kernel.org, kernel-team@cloudflare.com,
Chris J Arges <carges@cloudflare.com>
Subject: [PATCH net-next v3 2/3] ipv6: hash uncached routes by device
Date: Thu, 17 Sep 2026 14:38:23 -0500 [thread overview]
Message-ID: <20260917-hash-bucket-route-lists-v3-2-30493a37b6eb@cloudflare.com> (raw)
In-Reply-To: <20260917-hash-bucket-route-lists-v3-0-30493a37b6eb@cloudflare.com>
rt6_uncached_list_flush_dev() currently walks every per-CPU uncached route
list for each device being removed. Hash uncached routes by their inet6
device so ordinary device teardown only visits the matching bucket on each
CPU.
ip6_rt_get_dev_rcu() can return loopback or an L3 master while rt6i_idev
still refers to the original interface. Key routes by rt6i_idev->dev when
available and fall back to dst_dev(). Ordinary devices then require one
bucket scan. Because loopback and L3 masters can instead be referenced by
dst_dev(), scan all buckets when one of those devices is removed.
This avoids growing struct rt6_info while filtering most unrelated routes
from ordinary device teardown.
The table has 64 buckets and costs approximately 1.5 KiB per possible CPU
on x86-64.
Signed-off-by: Chris J Arges <carges@cloudflare.com>
---
net/ipv6/route.c | 101 +++++++++++++++++++++++++++++++++++++------------------
1 file changed, 69 insertions(+), 32 deletions(-)
diff --git a/net/ipv6/route.c b/net/ipv6/route.c
index 7535b09068a0..cda81e91be65 100644
--- a/net/ipv6/route.c
+++ b/net/ipv6/route.c
@@ -40,6 +40,7 @@
#include <linux/seq_file.h>
#include <linux/nsproxy.h>
#include <linux/slab.h>
+#include <linux/hash.h>
#include <linux/jhash.h>
#include <linux/siphash.h>
#include <net/net_namespace.h>
@@ -133,11 +134,23 @@ struct uncached_list {
struct list_head head;
};
-static DEFINE_PER_CPU_ALIGNED(struct uncached_list, rt6_uncached_list);
+#define RT6_UNCACHED_HASH_BITS 6
+#define RT6_UNCACHED_HASH_SIZE BIT(RT6_UNCACHED_HASH_BITS)
+
+struct rt6_uncached_table {
+ struct uncached_list buckets[RT6_UNCACHED_HASH_SIZE];
+};
+
+static DEFINE_PER_CPU_ALIGNED(struct rt6_uncached_table, rt6_uncached_table);
void rt6_uncached_list_add(struct rt6_info *rt)
{
- struct uncached_list *ul = raw_cpu_ptr(&rt6_uncached_list);
+ struct rt6_uncached_table *table = raw_cpu_ptr(&rt6_uncached_table);
+ struct uncached_list *ul;
+ struct net_device *dev;
+
+ dev = rt->rt6i_idev ? rt->rt6i_idev->dev : dst_dev(&rt->dst);
+ ul = &table->buckets[hash_ptr(dev, RT6_UNCACHED_HASH_BITS)];
rt->dst.rt_uncached_list = ul;
@@ -157,40 +170,58 @@ void rt6_uncached_list_del(struct rt6_info *rt)
}
}
+static void rt6_uncached_list_flush(struct uncached_list *ul,
+ struct net_device *dev)
+{
+ struct rt6_info *rt, *safe;
+
+ if (list_empty(&ul->head))
+ return;
+
+ spin_lock_bh(&ul->lock);
+ list_for_each_entry_safe(rt, safe, &ul->head, dst.rt_uncached) {
+ struct net_device *rt_dev = dst_dev(&rt->dst);
+ struct inet6_dev *rt_idev = rt->rt6i_idev;
+ bool handled = false;
+
+ if (rt_idev && rt_idev->dev == dev) {
+ rt->rt6i_idev = in6_dev_get(blackhole_netdev);
+ in6_dev_put(rt_idev);
+ handled = true;
+ }
+
+ if (rt_dev == dev) {
+ rcu_assign_pointer(rt->dst.dev_rcu, blackhole_netdev);
+ netdev_ref_replace(rt_dev, blackhole_netdev,
+ &rt->dst.dev_tracker, GFP_ATOMIC);
+ handled = true;
+ }
+ if (handled)
+ list_del_init(&rt->dst.rt_uncached);
+ }
+ spin_unlock_bh(&ul->lock);
+}
+
static void rt6_uncached_list_flush_dev(struct net_device *dev)
{
+ bool scan_all = dev->flags & IFF_LOOPBACK || netif_is_l3_master(dev);
int cpu;
for_each_possible_cpu(cpu) {
- struct uncached_list *ul = per_cpu_ptr(&rt6_uncached_list, cpu);
- struct rt6_info *rt, *safe;
-
- if (list_empty(&ul->head))
+ struct rt6_uncached_table *table;
+ struct uncached_list *ul;
+ int bucket;
+
+ table = per_cpu_ptr(&rt6_uncached_table, cpu);
+ if (!scan_all) {
+ ul = &table->buckets[hash_ptr(dev,
+ RT6_UNCACHED_HASH_BITS)];
+ rt6_uncached_list_flush(ul, dev);
continue;
-
- spin_lock_bh(&ul->lock);
- list_for_each_entry_safe(rt, safe, &ul->head, dst.rt_uncached) {
- struct inet6_dev *rt_idev = rt->rt6i_idev;
- struct net_device *rt_dev = rt->dst.dev;
- bool handled = false;
-
- if (rt_idev && rt_idev->dev == dev) {
- rt->rt6i_idev = in6_dev_get(blackhole_netdev);
- in6_dev_put(rt_idev);
- handled = true;
- }
-
- if (rt_dev == dev) {
- rt->dst.dev = blackhole_netdev;
- netdev_ref_replace(rt_dev, blackhole_netdev,
- &rt->dst.dev_tracker,
- GFP_ATOMIC);
- handled = true;
- }
- if (handled)
- list_del_init(&rt->dst.rt_uncached);
}
- spin_unlock_bh(&ul->lock);
+
+ for (bucket = 0; bucket < RT6_UNCACHED_HASH_SIZE; bucket++)
+ rt6_uncached_list_flush(&table->buckets[bucket], dev);
}
}
@@ -6987,10 +7018,16 @@ int __init ip6_route_init(void)
#endif
for_each_possible_cpu(cpu) {
- struct uncached_list *ul = per_cpu_ptr(&rt6_uncached_list, cpu);
+ struct rt6_uncached_table *table;
+ int bucket;
+
+ table = per_cpu_ptr(&rt6_uncached_table, cpu);
+ for (bucket = 0; bucket < RT6_UNCACHED_HASH_SIZE; bucket++) {
+ struct uncached_list *ul = &table->buckets[bucket];
- INIT_LIST_HEAD(&ul->head);
- spin_lock_init(&ul->lock);
+ INIT_LIST_HEAD(&ul->head);
+ spin_lock_init(&ul->lock);
+ }
}
out:
--
2.43.0
next prev parent reply other threads:[~2026-09-17 19:38 UTC|newest]
Thread overview: 11+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-17 19:38 [PATCH net-next v3 0/3] net: hash uncached route lists by device Chris J Arges
2026-09-17 19:38 ` [PATCH net-next v3 1/3] ipv4: hash uncached routes " Chris J Arges
2026-09-17 19:38 ` Chris J Arges [this message]
2026-09-21 20:20 ` [PATCH net-next v3 2/3] ipv6: " netdev-bot+sashiko
2026-09-17 19:38 ` [PATCH net-next v3 3/3] selftests: net: cover IPv6 uncached route device mismatch Chris J Arges
2026-09-21 20:20 ` netdev-bot+sashiko
2026-09-17 22:10 ` [PATCH net-next v3 0/3] net: hash uncached route lists by device Kuniyuki Iwashima
2026-09-18 0:38 ` Chris Arges
2026-09-18 4:19 ` Kuniyuki Iwashima
2026-09-18 18:24 ` Chris Arges
2026-09-18 18:41 ` Kuniyuki Iwashima
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260917-hash-bucket-route-lists-v3-2-30493a37b6eb@cloudflare.com \
--to=carges@cloudflare.com \
--cc=davem@davemloft.net \
--cc=dsahern@kernel.org \
--cc=edumazet@google.com \
--cc=horms@kernel.org \
--cc=idosch@nvidia.com \
--cc=kernel-team@cloudflare.com \
--cc=kuba@kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
--cc=shuah@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox