[PATCH 1/3] cephfs: Use d_alloc_trylock() in ceph_readdir_prepopulate()

From: NeilBrown

Date: Mon Sep 28 2026 - 22:56:49 EST


From: NeilBrown <neil@xxxxxxxxxx>

cephfs uses the results of readdir to prime the dcache.
ceph_readdir() is called with an exclusive lock on the directory
so it currently cannot race with any lookup which allocates a dentry.
So it is safe to simply d_alloc() a new dentry and then d_splice_alias()
to install it in the dcache.

After we lift d_alloc_parallel() out of the parent lock this won't be
safe.

The safe interface to use here is d_alloc_trylock() which will handle
any races and can fail if there is a concurrent lookup which is working
on an in-look dentry. In the rare case that this does fail there is
little cost in simply skipping the priming of the dcache for this name -
some other thread which owns the dentry will fill those details in soon
enough.

So change to use d_alloc_trylock() and handle -EWOULDBLOCK. Also use
QSTR_LEN() to initialise dname. full_name_hash() and d_lookup() are no
longer needed as d_alloc_trylock() includes all that.
Because we use d_alloc_trylock() we need to call d_lookup_done() to
ensure the dentry gets unlocked.

Signed-off-by: NeilBrown <neil@xxxxxxxxxx>
---
fs/ceph/inode.c | 38 +++++++++++++++++---------------------
1 file changed, 17 insertions(+), 21 deletions(-)

diff --git a/fs/ceph/inode.c b/fs/ceph/inode.c
index d52e2b389e0b..74a107b6ef3c 100644
--- a/fs/ceph/inode.c
+++ b/fs/ceph/inode.c
@@ -2069,9 +2069,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
struct ceph_mds_reply_dir_entry *rde = rinfo->dir_entries + i;
struct ceph_vino tvino;

- dname.name = rde->name;
- dname.len = rde->name_len;
- dname.hash = full_name_hash(parent, dname.name, dname.len);
+ dname = QSTR_LEN(rde->name, rde->name_len);

tvino.ino = le64_to_cpu(rde->inode.in->ino);
tvino.snap = le64_to_cpu(rde->inode.in->snapid);
@@ -2087,24 +2085,20 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
}

retry_lookup:
- dn = d_lookup(parent, &dname);
- doutc(cl, "d_lookup on parent=%p name=%.*s got %p\n",
+ dn = d_alloc_trylock(parent, &dname);
+ doutc(cl, "d_alloc_trylock on parent=%p name=%.*s got %p\n",
parent, dname.len, dname.name, dn);
-
- if (!dn) {
- dn = d_alloc(parent, &dname);
- doutc(cl, "d_alloc %p '%.*s' = %p\n", parent,
- dname.len, dname.name, dn);
- if (!dn) {
- doutc(cl, "d_alloc badness\n");
- err = -ENOMEM;
- goto out;
- }
- if (rde->is_nokey) {
- spin_lock(&dn->d_lock);
- dn->d_flags |= DCACHE_NOKEY_NAME;
- spin_unlock(&dn->d_lock);
- }
+ if (dn == ERR_PTR(-EWOULDBLOCK)) {
+ /* Some other thread is working on this name */
+ continue;
+ } else if (IS_ERR(dn)) {
+ doutc(cl, "d_alloc_trylock badness\n");
+ err = PTR_ERR(dn);
+ goto out;
+ } else if (d_in_lookup(dn) && rde->is_nokey) {
+ spin_lock(&dn->d_lock);
+ dn->d_flags |= DCACHE_NOKEY_NAME;
+ spin_unlock(&dn->d_lock);
} else if (d_really_is_positive(dn) &&
(ceph_ino(d_inode(dn)) != tvino.ino ||
ceph_snap(d_inode(dn)) != tvino.snap)) {
@@ -2133,6 +2127,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
in = ceph_get_inode(parent->d_sb, tvino, NULL);
if (IS_ERR(in)) {
doutc(cl, "new_inode badness\n");
+ d_lookup_done(dn);
d_drop(dn);
dput(dn);
err = PTR_ERR(in);
@@ -2159,7 +2154,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
if (inode_state_read_once(in) & I_NEW)
unlock_new_inode(in);

- if (d_really_is_negative(dn)) {
+ if (d_in_lookup(dn) || d_really_is_negative(dn)) {
if (ceph_security_xattr_deadlock(in)) {
doutc(cl, " skip splicing dn %p to inode %p"
" (security xattr deadlock)\n", dn, in);
@@ -2186,6 +2181,7 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
err = ret;
}
next_item:
+ d_lookup_done(dn);
dput(dn);
}
out:

base-commit: 3879f51857325da9bf3cfb073280257cd16ae067
--
2.50.0.107.gf914562f5916.dirty