[PATCH v4 sched_ext/for-7.3 18/40] sched_ext: RCU-protect the sub-sched tree's children/sibling lists

From: Tejun Heo

Date: Wed Jul 08 2026 - 17:28:24 EST


Future kfuncs need to walk descendants without scx_sched_lock. Make the
walker RCU-safe so that they can. A sub-sched's fields are initialized
before it is linked, so a walk that observes a linked node also observes its
setup. In-place changes after linking carry their own ordering.

Switch the children/sibling list ops to RCU and expand the descendant walker
to accept rcu_read_lock as a valid read-side context. Walkers that mutate
keep scx_sched_lock.

A sub-sched can be linked while an ancestor is bypassing, after the bypass
walk that propagates the depth has passed its parent. Bypass state is a
per-cpu flag plus a depth count and can't be established atomically at link
time, so refuse to link under a bypassing ancestor. Take scx_bypass_lock
across linking to check the parent's bypass state coherently.

v3: Reject linking under a bypassing ancestor instead of inheriting bypass_depth. (sashiko AI)
v2: Inherit bypass_depth before publishing @sch on the RCU sibling list.

Signed-off-by: Tejun Heo <tj@xxxxxxxxxx>
---
kernel/sched/ext/ext.c | 18 +++++++++++++++---
kernel/sched/ext/sub.c | 11 +++++++----
kernel/sched/ext/sub.h | 4 ++--
3 files changed, 24 insertions(+), 9 deletions(-)

diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index c5201553f566..46b80b31925e 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -5465,7 +5465,8 @@ s32 scx_link_sched(struct scx_sched *sch)
const char *err_msg = "";
s32 ret = 0;

- scoped_guard(raw_spinlock_irq, &scx_sched_lock) {
+ scoped_guard(raw_spinlock_irqsave, &scx_bypass_lock) /* for the parent bypass check */
+ scoped_guard(raw_spinlock, &scx_sched_lock) {
#ifdef CONFIG_EXT_SUB_SCHED
struct scx_sched *parent = scx_parent(sch);

@@ -5482,6 +5483,17 @@ s32 scx_link_sched(struct scx_sched *sch)
break;
}

+ /*
+ * Bypass state is spread across per-cpu flags and a
+ * depth count, so inheriting it is tricky and has no
+ * valid use case. Refuse it.
+ */
+ if (READ_ONCE(parent->bypass_depth)) {
+ err_msg = "parent bypassing";
+ ret = -EBUSY;
+ break;
+ }
+
ret = rhashtable_lookup_insert_fast(&scx_sched_hash,
&sch->hash_node, scx_sched_hash_params);
if (ret) {
@@ -5489,7 +5501,7 @@ s32 scx_link_sched(struct scx_sched *sch)
break;
}

- list_add_tail(&sch->sibling, &parent->children);
+ list_add_tail_rcu(&sch->sibling, &parent->children);
}
#endif /* CONFIG_EXT_SUB_SCHED */

@@ -5516,7 +5528,7 @@ void scx_unlink_sched(struct scx_sched *sch)
if (scx_parent(sch)) {
rhashtable_remove_fast(&scx_sched_hash, &sch->hash_node,
scx_sched_hash_params);
- list_del_init(&sch->sibling);
+ list_del_rcu(&sch->sibling);
}
#endif /* CONFIG_EXT_SUB_SCHED */
list_del_rcu(&sch->all);
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 3adec9343e46..5fe2f79064dc 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -35,21 +35,24 @@ struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sche
struct scx_sched *next;

lockdep_assert(lockdep_is_held(&scx_enable_mutex) ||
- lockdep_is_held(&scx_sched_lock));
+ lockdep_is_held(&scx_sched_lock) ||
+ rcu_read_lock_any_held());

/* if first iteration, visit @root */
if (!pos)
return root;

/* visit the first child if exists */
- next = list_first_entry_or_null(&pos->children, struct scx_sched, sibling);
+ next = list_first_or_null_rcu(&pos->children, struct scx_sched, sibling);
if (next)
return next;

/* no child, visit my or the closest ancestor's next sibling */
while (pos != root) {
- if (!list_is_last(&pos->sibling, &scx_parent(pos)->children))
- return list_next_entry(pos, sibling);
+ next = list_next_or_null_rcu(&scx_parent(pos)->children, &pos->sibling,
+ struct scx_sched, sibling);
+ if (next)
+ return next;
pos = scx_parent(pos);
}

diff --git a/kernel/sched/ext/sub.h b/kernel/sched/ext/sub.h
index 9fa6b5c8be23..e936867bc5c5 100644
--- a/kernel/sched/ext/sub.h
+++ b/kernel/sched/ext/sub.h
@@ -46,8 +46,8 @@ static inline s32 scx_alloc_pshards(struct scx_sched *sch) { return 0; }
* @root: sched to walk the descendants of
*
* Walk @root's descendants. @root is included in the iteration and the first
- * node to be visited. Must be called with either scx_enable_mutex or
- * scx_sched_lock held.
+ * node to be visited. Must be called with scx_enable_mutex, scx_sched_lock, or
+ * RCU read lock.
*/
#define scx_for_each_descendant_pre(pos, root) \
for ((pos) = scx_next_descendant_pre(NULL, (root)); (pos); \
--
2.54.0