From: Eric Dumazet <edumazet@google.com>
To: "David S . Miller" <davem@davemloft.net>,
Jakub Kicinski <kuba@kernel.org>,
Paolo Abeni <pabeni@redhat.com>
Cc: Simon Horman <horms@kernel.org>,
Kuniyuki Iwashima <kuniyu@google.com>,
netdev@vger.kernel.org, eric.dumazet@gmail.com,
Eric Dumazet <edumazet@google.com>
Subject: [PATCH v6 net-next 3/8] vxlan: vnifilter: signal interrupted RTM_GETTUNNEL dumps
Date: Tue, 22 Sep 2026 18:10:57 +0000 [thread overview]
Message-ID: <20260922181102.3989489-4-edumazet@google.com> (raw)
In-Reply-To: <20260922181102.3989489-1-edumazet@google.com>
vxlan_vnifilter_dump() walks all vxlan devices of a netns, and for each
one walks vg->vni_list. Both cursors are plain ordinals stored in
cb->args[0] and cb->args[1], and RTNL is released between dump skbs.
A concurrent __vxlan_vni_add_list(), which inserts sorted by VNI,
__vxlan_vni_del_list(), or an in-place group update in
vxlan_vni_update() (which splits or merges coalesced VNI ranges) shifts
the second cursor, duplicating or skipping entries. Unregistering a
vxlan device shifts the first one, and the partially dumped device is
then skipped altogether by the "if (idx < s_idx)" test, silently losing
the rest of its VNIs.
Add a per-netns generation counter, bumped whenever the set of vxlan
devices in the netns or any vni_list changes, and feed it to
nl_dump_check_consistent() so that user space gets NLM_F_DUMP_INTR and
can retry, as vxlan_mdb_dump() already does.
The counter is keyed on dev_net(vxlan->dev) rather than vxlan->net,
because the dump enumerates devices with for_each_netdev_rcu() in the
netns the netdevice lives in, which differs from the packet i/o netns
when the device was created with a separate link netns.
Fixes: f9c4bb0b245c ("vxlan: vni filtering support on collect metadata device")
Signed-off-by: Eric Dumazet <edumazet@google.com>
---
drivers/net/vxlan/vxlan_core.c | 4 ++++
drivers/net/vxlan/vxlan_private.h | 9 ++++++++
drivers/net/vxlan/vxlan_vnifilter.c | 35 ++++++++++++++++++++++++-----
3 files changed, 42 insertions(+), 6 deletions(-)
diff --git a/drivers/net/vxlan/vxlan_core.c b/drivers/net/vxlan/vxlan_core.c
index a2cede8b082ab90519c029c52e3cff9e018c2418..93a38e1b609a87bfc94865a462d5d5370fc02914 100644
--- a/drivers/net/vxlan/vxlan_core.c
+++ b/drivers/net/vxlan/vxlan_core.c
@@ -4771,6 +4771,10 @@ static int vxlan_netdevice_event(struct notifier_block *unused,
struct net_device *dev = netdev_notifier_info_to_dev(ptr);
struct vxlan_net *vn = net_generic(dev_net(dev), vxlan_net_id);
+ if ((event == NETDEV_REGISTER || event == NETDEV_UNREGISTER) &&
+ netif_is_vxlan(dev))
+ vxlan_vnifilter_seq_inc(dev_net(dev));
+
if (event == NETDEV_UNREGISTER)
vxlan_handle_lowerdev_unregister(vn, dev);
else if (event == NETDEV_UDP_TUNNEL_PUSH_INFO)
diff --git a/drivers/net/vxlan/vxlan_private.h b/drivers/net/vxlan/vxlan_private.h
index e4ceca925bde7bea909bd691fa0f4ebfebdefbd3..74d713717cd5da267fc0aef9812593b5ed3ef37e 100644
--- a/drivers/net/vxlan/vxlan_private.h
+++ b/drivers/net/vxlan/vxlan_private.h
@@ -22,6 +22,8 @@ struct vxlan_net {
/* sock_list is protected by rtnl lock */
struct hlist_head sock_list[PORT_HASH_SIZE];
struct notifier_block nexthop_notifier_block;
+ /* Generation counter for RTM_GETTUNNEL dumps */
+ atomic_t vnifilter_seq;
};
struct vxlan_fdb_key {
@@ -177,6 +179,13 @@ vxlan_vnifilter_lookup(struct vxlan_dev *vxlan, __be32 vni)
vxlan_vni_rht_params);
}
+static inline void vxlan_vnifilter_seq_inc(const struct net *net)
+{
+ struct vxlan_net *vn = net_generic(net, vxlan_net_id);
+
+ atomic_inc(&vn->vnifilter_seq);
+}
+
/* vxlan_core.c */
int vxlan_fdb_create(struct vxlan_dev *vxlan,
const u8 *mac, union vxlan_addr *ip,
diff --git a/drivers/net/vxlan/vxlan_vnifilter.c b/drivers/net/vxlan/vxlan_vnifilter.c
index 0a04e8875f7dc4da66a01de8d0e99f82c75c1fdd..181a7614be5cdba41ac4281cebb17b93d99d15df 100644
--- a/drivers/net/vxlan/vxlan_vnifilter.c
+++ b/drivers/net/vxlan/vxlan_vnifilter.c
@@ -410,9 +410,22 @@ static int vxlan_vnifilter_dump_dev(const struct net_device *dev,
nlmsg_end(skb, nlh);
+ nl_dump_check_consistent(cb, nlh);
+
return err;
}
+static u32 vxlan_vnifilter_base_seq(const struct net *net)
+{
+ const struct vxlan_net *vn = net_generic(net, vxlan_net_id);
+ u32 res = atomic_read(&vn->vnifilter_seq);
+
+ /* Must not return 0 (see nl_dump_check_consistent()) */
+ if (!res)
+ res = 0x80000000;
+ return res;
+}
+
static int vxlan_vnifilter_dump(struct sk_buff *skb, struct netlink_callback *cb)
{
int idx = 0, err = 0, s_idx = cb->args[0];
@@ -432,6 +445,9 @@ static int vxlan_vnifilter_dump(struct sk_buff *skb, struct netlink_callback *cb
}
rcu_read_lock();
+
+ cb->seq = vxlan_vnifilter_base_seq(net);
+
if (tmsg->ifindex) {
dev = dev_get_by_index_rcu(net, tmsg->ifindex);
if (!dev) {
@@ -571,8 +587,11 @@ static int vxlan_vni_update_group(struct vxlan_dev *vxlan,
if (ret)
goto out;
- if (group)
+ if (group) {
memcpy(&vninode->remote_ip, group, sizeof(vninode->remote_ip));
+ if (!create)
+ vxlan_vnifilter_seq_inc(dev_net(vxlan->dev));
+ }
if (vxlan->dev->flags & IFF_UP) {
if (vxlan_addr_multicast(&old_remote_ip) &&
@@ -719,7 +738,8 @@ static int vxlan_vni_update(struct vxlan_dev *vxlan,
return 0;
}
-static void __vxlan_vni_add_list(struct vxlan_vni_group *vg,
+static void __vxlan_vni_add_list(struct vxlan_dev *vxlan,
+ struct vxlan_vni_group *vg,
struct vxlan_vni_node *v)
{
struct list_head *headp, *hpos;
@@ -735,13 +755,16 @@ static void __vxlan_vni_add_list(struct vxlan_vni_group *vg,
}
list_add_rcu(&v->vlist, hpos);
vg->num_vnis++;
+ vxlan_vnifilter_seq_inc(dev_net(vxlan->dev));
}
-static void __vxlan_vni_del_list(struct vxlan_vni_group *vg,
+static void __vxlan_vni_del_list(struct vxlan_dev *vxlan,
+ struct vxlan_vni_group *vg,
struct vxlan_vni_node *v)
{
list_del_rcu(&v->vlist);
vg->num_vnis--;
+ vxlan_vnifilter_seq_inc(dev_net(vxlan->dev));
}
static struct vxlan_vni_node *vxlan_vni_alloc(struct vxlan_dev *vxlan,
@@ -803,7 +826,7 @@ static int vxlan_vni_add(struct vxlan_dev *vxlan,
return err;
}
- __vxlan_vni_add_list(vg, vninode);
+ __vxlan_vni_add_list(vxlan, vg, vninode);
if (vxlan->dev->flags & IFF_UP)
vxlan_vs_add_del_vninode(vxlan, vninode, false);
@@ -849,7 +872,7 @@ static int vxlan_vni_del(struct vxlan_dev *vxlan,
if (err)
goto out;
- __vxlan_vni_del_list(vg, vninode);
+ __vxlan_vni_del_list(vxlan, vg, vninode);
vxlan_vnifilter_notify(vxlan, vninode, RTM_DELTUNNEL);
@@ -963,7 +986,7 @@ void vxlan_vnigroup_uninit(struct vxlan_dev *vxlan)
#if IS_ENABLED(CONFIG_IPV6)
hlist_del_init_rcu(&v->hlist6.hlist);
#endif
- __vxlan_vni_del_list(vg, v);
+ __vxlan_vni_del_list(vxlan, vg, v);
vxlan_vnifilter_notify(vxlan, v, RTM_DELTUNNEL);
call_rcu(&v->rcu, vxlan_vni_node_rcu_free);
}
--
2.55.0.1082.g2b9226bbc0-goog
next prev parent reply other threads:[~2026-09-22 18:11 UTC|newest]
Thread overview: 15+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-22 18:10 [PATCH v6 net-next 0/8] vxlan: convert configuration to RCU and enable lockless dumps Eric Dumazet
2026-09-22 18:10 ` [PATCH v6 net-next 1/8] vxlan: update default fdb entries when the lower device changes Eric Dumazet
2026-09-24 0:11 ` netdev-bot+sashiko
2026-09-22 18:10 ` [PATCH v6 net-next 2/8] vxlan: vnifilter: use list_for_each_entry_rcu() in vxlan_vnifilter_dump_dev() Eric Dumazet
2026-09-24 0:11 ` netdev-bot+sashiko
2026-09-22 18:10 ` Eric Dumazet [this message]
2026-09-24 0:11 ` [PATCH v6 net-next 3/8] vxlan: vnifilter: signal interrupted RTM_GETTUNNEL dumps netdev-bot+sashiko
2026-09-22 18:10 ` [PATCH v6 net-next 4/8] vxlan: pass vxlan_config pointer to helper functions Eric Dumazet
2026-09-22 18:10 ` [PATCH v6 net-next 5/8] vxlan: move VXLAN_F_MDB to struct vxlan_dev flags Eric Dumazet
2026-09-22 18:11 ` [PATCH v6 net-next 6/8] vxlan: convert configuration to RCU protection Eric Dumazet
2026-09-22 18:11 ` [PATCH v6 net-next 7/8] vxlan: remove default_dst and use vxlan_config and lowerdev Eric Dumazet
2026-09-24 0:11 ` netdev-bot+sashiko
2026-09-22 18:11 ` [PATCH v6 net-next 8/8] vxlan: no longer rely on RTNL in vxlan_fill_info() Eric Dumazet
2026-09-22 18:20 ` [PATCH v6 net-next 0/8] vxlan: convert configuration to RCU and enable lockless dumps Jakub Kicinski
2026-09-28 23:50 ` patchwork-bot+netdevbpf
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260922181102.3989489-4-edumazet@google.com \
--to=edumazet@google.com \
--cc=davem@davemloft.net \
--cc=eric.dumazet@gmail.com \
--cc=horms@kernel.org \
--cc=kuba@kernel.org \
--cc=kuniyu@google.com \
--cc=netdev@vger.kernel.org \
--cc=pabeni@redhat.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox