Linux NFS development
 help / color / mirror / Atom feed
From: Benjamin Coddington <ben.coddington@hammerspace.com>
To: Trond Myklebust <trondmy@kernel.org>, Anna Schumaker <anna@kernel.org>
Cc: linux-nfs@vger.kernel.org,
	Jonathan Curley <jcurley@purestorage.com>,
	Mike Snitzer <snitzer@kernel.org>,
	Jeff Layton <jlayton@kernel.org>,
	Junrui Luo <moonafterrain@outlook.com>
Subject: [PATCH v4 22/24] NFSv4/pnfs: Re-home the data-server cache onto hash buckets
Date: Tue, 15 Sep 2026 08:22:24 -0400	[thread overview]
Message-ID: <91fcf4f64186fb459e3af9c83fe5344cfab6dfae.1789474702.git.bcodding@hammerspace.com> (raw)
In-Reply-To: <cover.1789474702.git.bcodding@hammerspace.com>

The per-net data-server cache is a single list, and every
GETDEVICEINFO decode walks all of it looking for a match, so filling
the cache costs O(n^2) in the number of data servers -- which a
striping mount does in one burst, at the same scale the deviceid
cache was just sized for.  Key it by the DS address set instead.

This patch is the mechanical half: nfs4_pnfs_ds.ds_node becomes an
hlist_node, netns init and teardown cover every bucket, and removal
uses hlist_del_init (which needs no bucket reference).  Insertion
still targets bucket 0 and lookup still scans every entry, so
behavior is unchanged; the key comes next.  Splitting it this way
keeps a bisect able to tell a list-conversion bug from a hash-key
bug.

The buckets live in struct nfs_net, so this costs 2KB per network
namespace on 64-bit, paid once nfs.ko is loaded whether or not that
namespace ever mounts NFS.

Assisted-by: Claude:claude-fable-5
Signed-off-by: Benjamin Coddington <bcodding@hammerspace.com>
---
 fs/nfs/client.c   |  6 ++++--
 fs/nfs/netns.h    |  5 ++++-
 fs/nfs/pnfs.h     |  2 +-
 fs/nfs/pnfs_nfs.c | 19 +++++++++++--------
 4 files changed, 20 insertions(+), 12 deletions(-)

diff --git a/fs/nfs/client.c b/fs/nfs/client.c
index 60386330aeec..dbb5131375f8 100644
--- a/fs/nfs/client.c
+++ b/fs/nfs/client.c
@@ -1295,7 +1295,8 @@ void nfs_clients_init(struct net *net)
 	INIT_LIST_HEAD(&nn->nfs_volume_list);
 #if IS_ENABLED(CONFIG_NFS_V4)
 	idr_init(&nn->cb_ident_idr);
-	INIT_LIST_HEAD(&nn->nfs4_data_server_cache);
+	for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++)
+		INIT_HLIST_HEAD(&nn->nfs4_data_server_cache[i]);
 	spin_lock_init(&nn->nfs4_data_server_lock);
 #endif /* CONFIG_NFS_V4 */
 	spin_lock_init(&nn->nfs_client_lock);
@@ -1315,7 +1316,8 @@ void nfs_clients_exit(struct net *net)
 	WARN_ON_ONCE(!list_empty(&nn->nfs_client_list));
 	WARN_ON_ONCE(!list_empty(&nn->nfs_volume_list));
 #if IS_ENABLED(CONFIG_NFS_V4)
-	WARN_ON_ONCE(!list_empty(&nn->nfs4_data_server_cache));
+	for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++)
+		WARN_ON_ONCE(!hlist_empty(&nn->nfs4_data_server_cache[i]));
 #endif /* CONFIG_NFS_V4 */
 }
 
diff --git a/fs/nfs/netns.h b/fs/nfs/netns.h
index 36658579100d..560fa95726b0 100644
--- a/fs/nfs/netns.h
+++ b/fs/nfs/netns.h
@@ -31,7 +31,10 @@ struct nfs_net {
 	unsigned short nfs_callback_tcpport;
 	unsigned short nfs_callback_tcpport6;
 	int cb_users[NFS4_MAX_MINOR_VERSION + 1];
-	struct list_head nfs4_data_server_cache;
+#define NFS4_DS_CACHE_HASH_BITS 8
+#define NFS4_DS_CACHE_HASH_SIZE (1 << NFS4_DS_CACHE_HASH_BITS)
+	/* every entry is still in bucket 0 until the key is added */
+	struct hlist_head nfs4_data_server_cache[NFS4_DS_CACHE_HASH_SIZE];
 	spinlock_t nfs4_data_server_lock;
 #endif /* CONFIG_NFS_V4 */
 	struct nfs_netns_client *nfs_client;
diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h
index 18ea4e8e0d85..3a1c012ba18e 100644
--- a/fs/nfs/pnfs.h
+++ b/fs/nfs/pnfs.h
@@ -57,7 +57,7 @@ struct nfs4_pnfs_ds_addr {
 };
 
 struct nfs4_pnfs_ds {
-	struct list_head	ds_node;  /* nfs4_pnfs_dev_hlist dev_dslist */
+	struct hlist_node	ds_node;  /* nfs_net nfs4_data_server_cache */
 	char			*ds_remotestr;	/* comma sep list of addrs */
 	struct list_head	ds_addrs;
 	const struct net	*ds_net;
diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c
index e0e3fc7414e6..f88a9784988a 100644
--- a/fs/nfs/pnfs_nfs.c
+++ b/fs/nfs/pnfs_nfs.c
@@ -604,7 +604,7 @@ _same_data_server_addrs_locked(const struct list_head *dsaddrs1,
 }
 
 /*
- * Lookup DS by addresses and NFS version.  nfs4_ds_cache_lock is held
+ * Lookup DS by addresses and NFS version.  nfs4_data_server_lock is held
  */
 static struct nfs4_pnfs_ds *
 _data_server_lookup_locked(const struct nfs_net *nn,
@@ -612,10 +612,13 @@ _data_server_lookup_locked(const struct nfs_net *nn,
 {
 	struct nfs4_pnfs_ds *ds;
 
-	list_for_each_entry(ds, &nn->nfs4_data_server_cache, ds_node)
-		if (ds->ds_version == version &&
-		    _same_data_server_addrs_locked(&ds->ds_addrs, dsaddrs))
-			return ds;
+	for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++)
+		hlist_for_each_entry(ds, &nn->nfs4_data_server_cache[i],
+				     ds_node)
+			if (ds->ds_version == version &&
+			    _same_data_server_addrs_locked(&ds->ds_addrs,
+							   dsaddrs))
+				return ds;
 	return NULL;
 }
 
@@ -666,7 +669,7 @@ void nfs4_pnfs_ds_put(struct nfs4_pnfs_ds *ds)
 	struct nfs_net *nn = net_generic(ds->ds_net, nfs_net_id);
 
 	if (refcount_dec_and_lock(&ds->ds_count, &nn->nfs4_data_server_lock)) {
-		list_del_init(&ds->ds_node);
+		hlist_del_init(&ds->ds_node);
 		spin_unlock(&nn->nfs4_data_server_lock);
 		destroy_ds(ds);
 	}
@@ -753,11 +756,11 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version,
 		list_splice_init(dsaddrs, &ds->ds_addrs);
 		ds->ds_remotestr = remotestr;
 		refcount_set(&ds->ds_count, 1);
-		INIT_LIST_HEAD(&ds->ds_node);
+		INIT_HLIST_NODE(&ds->ds_node);
 		ds->ds_net = net;
 		ds->ds_clp = NULL;
 		ds->ds_version = version;
-		list_add(&ds->ds_node, &nn->nfs4_data_server_cache);
+		hlist_add_head(&ds->ds_node, &nn->nfs4_data_server_cache[0]);
 		dprintk("%s add new data server %s\n", __func__,
 			ds->ds_remotestr);
 	} else {
-- 
2.53.0


  parent reply	other threads:[~2026-09-15 12:22 UTC|newest]

Thread overview: 26+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-15 12:22 [PATCH v4 00/24] NFS: flexfiles device notifications and caching for wide striped layouts Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 01/24] NFSv4/pnfs: Free the netid when draining a data-server address list Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 02/24] NFSv4/flexfiles: Use the full 64-bit stripe_unit Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 03/24] NFSv4/pnfs: bound the CB_NOTIFY_DEVICEID array count before allocating Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 04/24] pNFS: Fix CB_NOTIFY_DEVICEID CHANGE to consume ndc_immediate Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 05/24] NFSv4/flexfiles: Use the full 64-bit offset for read DS selection Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 06/24] NFSv4/flexfiles: Bound page coalescing on the absolute stripe offset Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 07/24] NFSv4/filelayout: Anchor page coalescing on pattern_offset Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 08/24] NFSv4/flexfiles: Reference the device node across DS setup Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 09/24] NFSv4/flexfiles: Carry the device node reference across each I/O Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 10/24] NFSv4/flexfiles: Hold a device node reference for layoutstats encoding Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 11/24] NFSv4/flexfiles: Make the pinned device node pointer RCU-managed Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 12/24] pNFS: Add a reresolve_deviceid layout driver hook Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 13/24] NFSv4/flexfiles: Implement in-place device re-resolve on CHANGE Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 14/24] NFSv4: Dispatch CB_NOTIFY_DEVICEID CHANGE to an in-place refresh Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 15/24] NFSv4/flexfiles: Honor ndc_immediate on CB_NOTIFY_DEVICEID CHANGE Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 16/24] pNFS: Discard a GETDEVICEINFO reply that raced a CHANGE notification Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 17/24] pNFS: Add deviceid reference query and collection walkers Benjamin Coddington
2026-09-15 17:29   ` Anna Schumaker
2026-09-15 12:22 ` [PATCH v4 18/24] NFSv4/pnfs: Recover revoked layouts on a deleted deviceID Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 19/24] NFSv4/pnfs: Confirm a deviceID delete via GETDEVICEINFO Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 20/24] NFSv4/pnfs: Dispatch CB_NOTIFY_DEVICEID DELETE to race recovery Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 21/24] NFSv4/pnfs: Grow the deviceid cache hash table Benjamin Coddington
2026-09-15 12:22 ` Benjamin Coddington [this message]
2026-09-15 12:22 ` [PATCH v4 23/24] NFSv4/pnfs: Key the data-server cache by its address set and version Benjamin Coddington
2026-09-15 12:22 ` [PATCH v4 24/24] NFSv4/flexfiles: Add a dataserver_nconnect cap Benjamin Coddington

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=91fcf4f64186fb459e3af9c83fe5344cfab6dfae.1789474702.git.bcodding@hammerspace.com \
    --to=ben.coddington@hammerspace.com \
    --cc=anna@kernel.org \
    --cc=jcurley@purestorage.com \
    --cc=jlayton@kernel.org \
    --cc=linux-nfs@vger.kernel.org \
    --cc=moonafterrain@outlook.com \
    --cc=snitzer@kernel.org \
    --cc=trondmy@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox