Linux filesystem development
 help / color / mirror / Atom feed
From: NeilBrown <neilb@ownmail.net>
To: Ilya Dryomov <idryomov@gmail.com>,
	Alex Markuze <amarkuze@redhat.com>,
	Viacheslav Dubeyko <slava@dubeyko.com>,
	Alexander Viro <viro@zeniv.linux.org.uk>,
	Christian Brauner <brauner@kernel.org>
Cc: Jeff Layton <jlayton@kernel.org>, Jan Kara <jack@suse.cz>,
	linux-fsdevel@vger.kernel.org, ceph-devel@vger.kernel.org,
	linux-kernel@vger.kernel.org
Subject: [PATCH 1/3] cephfs: Use d_alloc_trylock() in ceph_readdir_prepopulate()
Date: Tue, 29 Sep 2026 12:47:39 +1000	[thread overview]
Message-ID: <20260929025331.1436824-2-neilb@ownmail.net> (raw)
In-Reply-To: <20260929025331.1436824-1-neilb@ownmail.net>

From: NeilBrown <neil@brown.name>

cephfs uses the results of readdir to prime the dcache.
ceph_readdir() is called with an exclusive lock on the directory
so it currently cannot race with any lookup which allocates a dentry.
So it is safe to simply d_alloc() a new dentry and then d_splice_alias()
to install it in the dcache.

After we lift d_alloc_parallel() out of the parent lock this won't be
safe.

The safe interface to use here is d_alloc_trylock() which will handle
any races and can fail if there is a concurrent lookup which is working
on an in-look dentry.  In the rare case that this does fail there is
little cost in simply skipping the priming of the dcache for this name -
some other thread which owns the dentry will fill those details in soon
enough.

So change to use d_alloc_trylock() and handle -EWOULDBLOCK.  Also use
QSTR_LEN() to initialise dname.  full_name_hash() and d_lookup() are no
longer needed as d_alloc_trylock() includes all that.
Because we use d_alloc_trylock() we need to call d_lookup_done() to
ensure the dentry gets unlocked.

Signed-off-by: NeilBrown <neil@brown.name>
---
 fs/ceph/inode.c | 38 +++++++++++++++++---------------------
 1 file changed, 17 insertions(+), 21 deletions(-)

diff --git a/fs/ceph/inode.c b/fs/ceph/inode.c
index d52e2b389e0b..74a107b6ef3c 100644
--- a/fs/ceph/inode.c
+++ b/fs/ceph/inode.c
@@ -2069,9 +2069,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
 		struct ceph_mds_reply_dir_entry *rde = rinfo->dir_entries + i;
 		struct ceph_vino tvino;
 
-		dname.name = rde->name;
-		dname.len = rde->name_len;
-		dname.hash = full_name_hash(parent, dname.name, dname.len);
+		dname = QSTR_LEN(rde->name, rde->name_len);
 
 		tvino.ino = le64_to_cpu(rde->inode.in->ino);
 		tvino.snap = le64_to_cpu(rde->inode.in->snapid);
@@ -2087,24 +2085,20 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
 		}
 
 retry_lookup:
-		dn = d_lookup(parent, &dname);
-		doutc(cl, "d_lookup on parent=%p name=%.*s got %p\n",
+		dn = d_alloc_trylock(parent, &dname);
+		doutc(cl, "d_alloc_trylock on parent=%p name=%.*s got %p\n",
 		      parent, dname.len, dname.name, dn);
-
-		if (!dn) {
-			dn = d_alloc(parent, &dname);
-			doutc(cl, "d_alloc %p '%.*s' = %p\n", parent,
-			      dname.len, dname.name, dn);
-			if (!dn) {
-				doutc(cl, "d_alloc badness\n");
-				err = -ENOMEM;
-				goto out;
-			}
-			if (rde->is_nokey) {
-				spin_lock(&dn->d_lock);
-				dn->d_flags |= DCACHE_NOKEY_NAME;
-				spin_unlock(&dn->d_lock);
-			}
+		if (dn == ERR_PTR(-EWOULDBLOCK)) {
+			/* Some other thread is working on this name */
+			continue;
+		} else if (IS_ERR(dn)) {
+			doutc(cl, "d_alloc_trylock badness\n");
+			err = PTR_ERR(dn);
+			goto out;
+		} else if (d_in_lookup(dn) && rde->is_nokey) {
+			spin_lock(&dn->d_lock);
+			dn->d_flags |= DCACHE_NOKEY_NAME;
+			spin_unlock(&dn->d_lock);
 		} else if (d_really_is_positive(dn) &&
 			   (ceph_ino(d_inode(dn)) != tvino.ino ||
 			    ceph_snap(d_inode(dn)) != tvino.snap)) {
@@ -2133,6 +2127,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
 			in = ceph_get_inode(parent->d_sb, tvino, NULL);
 			if (IS_ERR(in)) {
 				doutc(cl, "new_inode badness\n");
+				d_lookup_done(dn);
 				d_drop(dn);
 				dput(dn);
 				err = PTR_ERR(in);
@@ -2159,7 +2154,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
 		if (inode_state_read_once(in) & I_NEW)
 			unlock_new_inode(in);
 
-		if (d_really_is_negative(dn)) {
+		if (d_in_lookup(dn) || d_really_is_negative(dn)) {
 			if (ceph_security_xattr_deadlock(in)) {
 				doutc(cl, " skip splicing dn %p to inode %p"
 				      " (security xattr deadlock)\n", dn, in);
@@ -2186,6 +2181,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
 				err = ret;
 		}
 next_item:
+		d_lookup_done(dn);
 		dput(dn);
 	}
 out:

base-commit: 3879f51857325da9bf3cfb073280257cd16ae067
-- 
2.50.0.107.gf914562f5916.dirty


  reply	other threads:[~2026-09-29  2:54 UTC|newest]

Thread overview: 4+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-29  2:47 [PATCH 0/3] cephfs: prepare for changes to VFS locking NeilBrown
2026-09-29  2:47 ` NeilBrown [this message]
2026-09-29  2:47 ` [PATCH 2/3] cephfs: Don't d_drop() before d_splice_alias() NeilBrown
2026-09-29  2:47 ` [PATCH 3/3] cephfs: stop using d_add() NeilBrown

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260929025331.1436824-2-neilb@ownmail.net \
    --to=neilb@ownmail.net \
    --cc=amarkuze@redhat.com \
    --cc=brauner@kernel.org \
    --cc=ceph-devel@vger.kernel.org \
    --cc=idryomov@gmail.com \
    --cc=jack@suse.cz \
    --cc=jlayton@kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=neil@brown.name \
    --cc=slava@dubeyko.com \
    --cc=viro@zeniv.linux.org.uk \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox