[PATCH 13/21] fhandle: decide the subtree check under mount_lock
From: Christian Brauner
Date: Fri Oct 02 2026 - 10:00:18 EST
may_decode_fh() lets a caller who is privileged over the mount
namespace of @root decode handles below @root->dentry as long as the
mount is mounted and no locked child covers something below it. The
three parts are read one after the other without a lock: is_mounted()
and capable_wrt_mount() read ->mnt_ns and has_locked_children() walks
->mnt_mounts under a mount_lock of its own.
Today the gaps are harmless. A lazy umount in between clears ->mnt_ns,
but the locked children stay attached to their unmounted parent, so the
walk still finds them and the decode is refused either way. The
following patches detach every unmounted mount from its parent. Then
an umount between is_mounted() and the walk makes the walk come back
empty and the caller decodes into what a locked child covered.
So take mount_lock once and answer all three questions under it.
is_mounted() is stable there, umount_tree() clears ->mnt_ns on the
write side, and a mounted mount still has its locked children on its
list. ns_capable() under the spinlock is fine, generic_permission()
calls it in RCU walk already. has_locked_children() loses its locking
wrapper, its other callers hold namespace_sem or mount_lock anyway.
Signed-off-by: Christian Brauner (Amutable) <brauner@xxxxxxxxxx>
---
fs/fhandle.c | 20 +++++++++++++++++---
fs/namespace.c | 14 ++++----------
2 files changed, 21 insertions(+), 13 deletions(-)
diff --git a/fs/fhandle.c b/fs/fhandle.c
index f8829231e3d7..d22f2e065677 100644
--- a/fs/fhandle.c
+++ b/fs/fhandle.c
@@ -298,6 +298,22 @@ static bool capable_wrt_mount(struct mount *mount)
return mnt_ns && ns_capable(mnt_ns->user_ns, CAP_SYS_ADMIN);
}
+/*
+ * Does the caller have an unobstructed way to everything below @root? Only
+ * if the mount is mounted, the caller is privileged over its mount namespace
+ * and no locked child covers something below @root->dentry. One answer from
+ * under mount_lock: an umount in between clears ->mnt_ns and takes the
+ * children off the mount, the locked ones too.
+ */
+static bool subtree_unobstructed(const struct path *root)
+{
+ struct mount *mnt = real_mount(root->mnt);
+
+ guard(mount_locked_reader)();
+ return is_mounted(root->mnt) && capable_wrt_mount(mnt) &&
+ !has_locked_children(mnt, root->dentry);
+}
+
static inline int may_decode_fh(struct handle_to_path_ctx *ctx,
unsigned int o_flags)
{
@@ -332,9 +348,7 @@ static inline int may_decode_fh(struct handle_to_path_ctx *ctx,
if (ns_capable(root->mnt->mnt_sb->s_user_ns, CAP_SYS_ADMIN))
ctx->flags = HANDLE_CHECK_PERMS;
- else if (is_mounted(root->mnt) &&
- capable_wrt_mount(real_mount(root->mnt)) &&
- !has_locked_children(real_mount(root->mnt), root->dentry))
+ else if (subtree_unobstructed(root))
ctx->flags = HANDLE_CHECK_PERMS | HANDLE_CHECK_SUBTREE;
else
return -EPERM;
diff --git a/fs/namespace.c b/fs/namespace.c
index bb0183ec2aaf..e116894c8667 100644
--- a/fs/namespace.c
+++ b/fs/namespace.c
@@ -2353,7 +2353,7 @@ void dissolve_on_fput(struct vfsmount *mnt)
}
/* locks: namespace_shared && pinned(mnt) || mount_locked_reader */
-static bool __has_locked_children(struct mount *mnt, struct dentry *dentry)
+bool has_locked_children(struct mount *mnt, struct dentry *dentry)
{
struct mount *child;
@@ -2367,12 +2367,6 @@ static bool __has_locked_children(struct mount *mnt, struct dentry *dentry)
return false;
}
-bool has_locked_children(struct mount *mnt, struct dentry *dentry)
-{
- guard(mount_locked_reader)();
- return __has_locked_children(mnt, dentry);
-}
-
/* locks: namespace_shared && pinned(mnt) || mount_locked_reader */
static bool __has_children(struct mount *mnt, struct dentry *dentry)
{
@@ -2441,7 +2435,7 @@ struct vfsmount *clone_private_mount(const struct path *path)
if (!ns_capable(old_mnt->mnt_ns->user_ns, CAP_SYS_ADMIN))
return ERR_PTR(-EPERM);
- if (__has_locked_children(old_mnt, path->dentry))
+ if (has_locked_children(old_mnt, path->dentry))
return ERR_PTR(-EINVAL);
new_mnt = clone_mnt(old_mnt, path->dentry, CL_PRIVATE);
@@ -3056,7 +3050,7 @@ static struct mount *__do_loopback(const struct path *old_path,
if (recurse && !old->mnt_ns)
return ERR_PTR(-EINVAL);
- if (!recurse && __has_locked_children(old, old_path->dentry))
+ if (!recurse && has_locked_children(old, old_path->dentry))
return ERR_PTR(-EINVAL);
if (recurse)
@@ -3544,7 +3538,7 @@ static int do_set_group(const struct path *from_path, const struct path *to_path
return -EINVAL;
/* From mount should not have locked children in place of To's root */
- if (__has_locked_children(from, to->mnt.mnt_root))
+ if (has_locked_children(from, to->mnt.mnt_root))
return -EINVAL;
/* Setting sharing groups is only allowed on private mounts */
--
2.53.0