From: Benjamin Coddington <ben.coddington@hammerspace.com>
To: Trond Myklebust <trondmy@kernel.org>, Anna Schumaker <anna@kernel.org>
Cc: linux-nfs@vger.kernel.org, Mike Snitzer <snitzer@kernel.org>
Subject: [PATCH 2/2] NFSv4/pnfs: Let unused data server connections linger
Date: Fri, 18 Sep 2026 08:21:02 -0400 [thread overview]
Message-ID: <84d9205181ec20687f73baba489ec9a44da22dfe.1789733774.git.bcodding@hammerspace.com> (raw)
In-Reply-To: <cover.1789733774.git.bcodding@hammerspace.com>
A metadata server that does not support deviceid notifications makes
the client set NFS_DEVICEID_NOCACHE, and a NOCACHE deviceid node is
freed as soon as the last layout referring to it is returned.
The next IO will then do a synchronous connect and NULL ping inline in
pg_init(), plus the remaining nconnect-1 transports reconnecting lazily as
RPCs land on them. With a wide stripe and a large nconnect this introduces
a hefty penalty for eeach layout cycle.
Allow the transports to linger. Keep the nfs4_pnfs_ds hashed once its last
user goes away, and let a per-entry delayed work tear it down if nothing
claims it within dataserver_linger seconds.
he linger defaults to 120 seconds and can be set to zero to restore
the previous behavior. sunrpc already disconnects an idle transport
after five minutes, so an entry that is never claimed again holds
objects and slot tables rather than open connections.
Assisted-by: Claude:claude-opus-5
Signed-off-by: Benjamin Coddington <bcodding@hammerspace.com>
---
fs/nfs/pnfs.c | 4 +-
fs/nfs/pnfs.h | 4 ++
fs/nfs/pnfs_nfs.c | 95 ++++++++++++++++++++++++++++++++++++++++++++++-
3 files changed, 101 insertions(+), 2 deletions(-)
diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c
index ade314de1738..b7a7fab68498 100644
--- a/fs/nfs/pnfs.c
+++ b/fs/nfs/pnfs.c
@@ -111,8 +111,10 @@ unset_pnfs_layoutdriver(struct nfs_server *nfss)
if (nfss->pnfs_curr_ld->clear_layoutdriver)
nfss->pnfs_curr_ld->clear_layoutdriver(nfss);
/* Decrement the MDS count. Purge the deviceid cache if zero */
- if (atomic_dec_and_test(&nfss->nfs_client->cl_mds_count))
+ if (atomic_dec_and_test(&nfss->nfs_client->cl_mds_count)) {
nfs4_deviceid_purge_client(nfss->nfs_client);
+ nfs4_pnfs_ds_reap_net(nfss->nfs_client->cl_net);
+ }
module_put(nfss->pnfs_curr_ld->owner);
}
nfss->pnfs_curr_ld = NULL;
diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h
index 5cde5db63d29..5ce1a0746e09 100644
--- a/fs/nfs/pnfs.h
+++ b/fs/nfs/pnfs.h
@@ -58,6 +58,7 @@ struct nfs4_pnfs_ds_addr {
struct nfs4_pnfs_ds {
struct hlist_node ds_node; /* nfs_net nfs4_data_server_cache */
+ struct hlist_node ds_tmpnode; /* for batched disposal */
char *ds_remotestr; /* comma sep list of addrs */
struct list_head ds_addrs;
const struct net *ds_net;
@@ -66,6 +67,8 @@ struct nfs4_pnfs_ds {
u32 ds_version; /* cache key, with ds_addrs */
unsigned long ds_state;
#define NFS4DS_CONNECTING 0 /* ds is establishing connection */
+ unsigned long ds_idle; /* jiffies of the last put */
+ struct delayed_work ds_reaper;
};
struct pnfs_layout_segment {
@@ -498,6 +501,7 @@ int pnfs_generic_commit_pagelist(struct inode *inode,
int pnfs_generic_scan_commit_lists(struct nfs_commit_info *cinfo, int max);
void pnfs_generic_write_commit_done(struct rpc_task *task, void *data);
void nfs4_pnfs_ds_put(struct nfs4_pnfs_ds *ds);
+void nfs4_pnfs_ds_reap_net(const struct net *net);
struct nfs4_pnfs_ds *nfs4_pnfs_ds_add(const struct net *net,
struct list_head *dsaddrs,
u32 version, gfp_t gfp_flags);
diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c
index 6d6a4d05d13e..14d419b8bfc7 100644
--- a/fs/nfs/pnfs_nfs.c
+++ b/fs/nfs/pnfs_nfs.c
@@ -695,7 +695,22 @@ void nfs4_pnfs_ds_addr_list_free(struct list_head *dsaddrs)
}
EXPORT_SYMBOL_GPL(nfs4_pnfs_ds_addr_list_free);
-static void destroy_ds(struct nfs4_pnfs_ds *ds)
+static unsigned int dataserver_linger = 120;
+module_param(dataserver_linger, uint, 0644);
+MODULE_PARM_DESC(dataserver_linger,
+ "Seconds an unused pNFS data server connection is cached (0 disables)");
+
+/* Clamped so that a large value cannot overflow the jiffies conversion */
+static unsigned long nfs4_ds_linger(void)
+{
+ unsigned int secs = READ_ONCE(dataserver_linger);
+
+ if (!secs)
+ return 0;
+ return min(secs, 3600U) * HZ;
+}
+
+static void __destroy_ds(struct nfs4_pnfs_ds *ds)
{
dprintk("--> %s\n", __func__);
ifdebug(FACILITY)
@@ -709,9 +724,79 @@ static void destroy_ds(struct nfs4_pnfs_ds *ds)
kfree(ds);
}
+static void destroy_ds(struct nfs4_pnfs_ds *ds)
+{
+ cancel_delayed_work_sync(&ds->ds_reaper);
+ __destroy_ds(ds);
+}
+
+static void nfs4_pnfs_ds_reap_work(struct work_struct *work)
+{
+ struct nfs4_pnfs_ds *ds = container_of(to_delayed_work(work),
+ struct nfs4_pnfs_ds, ds_reaper);
+ struct nfs_net *nn = net_generic(ds->ds_net, nfs_net_id);
+ unsigned long linger, expires;
+
+ spin_lock(&nn->nfs4_data_server_lock);
+ if (hlist_unhashed(&ds->ds_node) ||
+ refcount_read(&ds->ds_count) > 1) {
+ spin_unlock(&nn->nfs4_data_server_lock);
+ return;
+ }
+ linger = nfs4_ds_linger();
+ expires = ds->ds_idle + linger;
+ /*
+ * Was the DS revived and put again after we were queued?
+ */
+ if (linger && time_before(jiffies, expires)) {
+ queue_delayed_work(nfsiod_workqueue, &ds->ds_reaper,
+ expires - jiffies);
+ spin_unlock(&nn->nfs4_data_server_lock);
+ return;
+ }
+ hlist_del_init(&ds->ds_node);
+ spin_unlock(&nn->nfs4_data_server_lock);
+
+ if (refcount_dec_and_test(&ds->ds_count))
+ __destroy_ds(ds);
+}
+
+/*
+ * Tear down every unused data server in @net immediately
+ */
+void nfs4_pnfs_ds_reap_net(const struct net *net)
+{
+ struct nfs_net *nn = net_generic(net, nfs_net_id);
+ struct nfs4_pnfs_ds *ds;
+ struct hlist_node *tmp;
+ HLIST_HEAD(dispose);
+ int i;
+
+ spin_lock(&nn->nfs4_data_server_lock);
+ for (i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++) {
+ hlist_for_each_entry_safe(ds, tmp,
+ &nn->nfs4_data_server_cache[i],
+ ds_node) {
+ if (refcount_read(&ds->ds_count) > 1)
+ continue;
+ hlist_del_init(&ds->ds_node);
+ hlist_add_head(&ds->ds_tmpnode, &dispose);
+ }
+ }
+ spin_unlock(&nn->nfs4_data_server_lock);
+
+ while (!hlist_empty(&dispose)) {
+ ds = hlist_entry(dispose.first, struct nfs4_pnfs_ds, ds_tmpnode);
+ hlist_del(&ds->ds_tmpnode);
+ if (refcount_dec_and_test(&ds->ds_count))
+ destroy_ds(ds);
+ }
+}
+
void nfs4_pnfs_ds_put(struct nfs4_pnfs_ds *ds)
{
struct nfs_net *nn = net_generic(ds->ds_net, nfs_net_id);
+ unsigned long linger = nfs4_ds_linger();
spin_lock(&nn->nfs4_data_server_lock);
refcount_dec(&ds->ds_count);
@@ -719,6 +804,13 @@ void nfs4_pnfs_ds_put(struct nfs4_pnfs_ds *ds)
spin_unlock(&nn->nfs4_data_server_lock);
return;
}
+
+ if (linger) {
+ ds->ds_idle = jiffies;
+ queue_delayed_work(nfsiod_workqueue, &ds->ds_reaper, linger);
+ spin_unlock(&nn->nfs4_data_server_lock);
+ return;
+ }
hlist_del_init(&ds->ds_node);
spin_unlock(&nn->nfs4_data_server_lock);
@@ -815,6 +907,7 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version,
ds->ds_net = net;
ds->ds_clp = NULL;
ds->ds_version = version;
+ INIT_DELAYED_WORK(&ds->ds_reaper, nfs4_pnfs_ds_reap_work);
hlist_add_head(&ds->ds_node, bucket);
dprintk("%s add new data server %s\n", __func__,
ds->ds_remotestr);
--
2.53.0
prev parent reply other threads:[~2026-09-18 12:21 UTC|newest]
Thread overview: 3+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-18 12:21 [PATCH 0/2] NFSv4/pnfs: Keep data server connections across deviceid eviction Benjamin Coddington
2026-09-18 12:21 ` [PATCH 1/2] NFSv4/pnfs: Give the data server cache its own reference Benjamin Coddington
2026-09-18 12:21 ` Benjamin Coddington [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=84d9205181ec20687f73baba489ec9a44da22dfe.1789733774.git.bcodding@hammerspace.com \
--to=ben.coddington@hammerspace.com \
--cc=anna@kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=snitzer@kernel.org \
--cc=trondmy@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox