[PATCH sched_ext/for-7.3 11/32] sched_ext: Defer scx_sched kobj sysfs add into the enable workfns

From: Tejun Heo

Date: Fri Jul 03 2026 - 04:17:19 EST


Split kobject_init_and_add() in scx_alloc_and_add_sched(): only
kobject_init() runs there. A new scx_sched_sysfs_add() helper does
kobject_add() (and creates sub_kset when the scheduler implements
ops.sub_attach), called by both enable workfns once @sch is linked and its
sysfs-visible state is initialized. Prep so a future caps attribute can rely
on @sch being fully built by the time it's sysfs-visible. Add early enough
that a stall later in enable still leaves sysfs inspectable.

Signed-off-by: Tejun Heo <tj@xxxxxxxxxx>
---
kernel/sched/ext/ext.c | 73 +++++++++++++++++++++++--------------
kernel/sched/ext/internal.h | 1 +
kernel/sched/ext/sub.c | 8 +++-
3 files changed, 53 insertions(+), 29 deletions(-)

diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 87a3fb9bb446..fcb8bf0d2422 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -5704,7 +5704,9 @@ static void scx_root_disable(struct scx_sched *sch)
if (sch->sub_kset)
kobject_del(&sch->sub_kset->kobj);
#endif
- kobject_del(&sch->kobj);
+ /* not added if enable failed before scx_sched_sysfs_add() */
+ if (sch->kobj.state_in_sysfs)
+ kobject_del(&sch->kobj);

free_kick_syncs();

@@ -6421,36 +6423,15 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
* disable. Released in scx_sched_free_rcu_work().
*/
kobject_get(&parent->kobj);
- ret = kobject_init_and_add(&sch->kobj, &scx_ktype,
- &parent->sub_kset->kobj,
- "sub-%llu", cgroup_id(cgrp));
- } else {
- ret = kobject_init_and_add(&sch->kobj, &scx_ktype, NULL, "root");
- }
-
- if (ret < 0) {
- RCU_INIT_POINTER(ops->priv, NULL);
- kobject_put(&sch->kobj);
- return ERR_PTR(ret);
- }
-
- if (ops->sub_attach) {
- sch->sub_kset = kset_create_and_add("sub", NULL, &sch->kobj);
- if (!sch->sub_kset) {
- RCU_INIT_POINTER(ops->priv, NULL);
- kobject_put(&sch->kobj);
- return ERR_PTR(-ENOMEM);
- }
- }
-#else /* CONFIG_EXT_SUB_SCHED */
- ret = kobject_init_and_add(&sch->kobj, &scx_ktype, NULL, "root");
- if (ret < 0) {
- RCU_INIT_POINTER(ops->priv, NULL);
- kobject_put(&sch->kobj);
- return ERR_PTR(ret);
}
#endif /* CONFIG_EXT_SUB_SCHED */

+ /*
+ * Init the kobj but don't add to sysfs yet. The enable path calls
+ * scx_sched_sysfs_add() once @sch's sysfs-visible state is initialized.
+ */
+ kobject_init(&sch->kobj, &scx_ktype);
+
/*
* Consume the arena_map ref bpf_scx_reg_cid() took. Defer to here so
* earlier failure paths leave cmd->arena_map set and bpf_scx_reg_cid
@@ -6508,6 +6489,36 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
return ERR_PTR(ret);
}

+/*
+ * Add @sch's kobject to sysfs, and create its sub_kset if the scheduler
+ * implements ops.sub_attach. Called by the enable workfns once @sch's
+ * sysfs-visible state is initialized.
+ */
+int scx_sched_sysfs_add(struct scx_sched *sch)
+{
+#ifdef CONFIG_EXT_SUB_SCHED
+ struct scx_sched *parent = scx_parent(sch);
+ int ret;
+
+ if (parent)
+ ret = kobject_add(&sch->kobj, &parent->sub_kset->kobj,
+ "sub-%llu", cgroup_id(sch_cgroup(sch)));
+ else
+ ret = kobject_add(&sch->kobj, NULL, "root");
+ if (ret < 0)
+ return ret;
+
+ if (sch->ops.sub_attach) {
+ sch->sub_kset = kset_create_and_add("sub", NULL, &sch->kobj);
+ if (!sch->sub_kset)
+ return -ENOMEM;
+ }
+ return 0;
+#else
+ return kobject_add(&sch->kobj, NULL, "root");
+#endif
+}
+
static int check_hotplug_seq(struct scx_sched *sch,
const struct sched_ext_ops *ops)
{
@@ -6730,6 +6741,12 @@ static void scx_root_enable_workfn(struct kthread_work *work)
sch->exit_info->flags |= SCX_EFLAG_INITIALIZED;
}

+ ret = scx_sched_sysfs_add(sch);
+ if (ret) {
+ cpus_read_unlock();
+ goto err_disable;
+ }
+
for (i = SCX_OPI_CPU_HOTPLUG_BEGIN; i < SCX_OPI_CPU_HOTPLUG_END; i++)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index ba5e9be0e3c3..7c6f4ed10cde 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -1689,6 +1689,7 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
struct cgroup *cgrp,
struct scx_sched *parent);
int scx_validate_ops(struct scx_sched *sch, const struct sched_ext_ops *ops);
+int scx_sched_sysfs_add(struct scx_sched *sch);

extern raw_spinlock_t scx_sched_lock;
extern struct mutex scx_enable_mutex;
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 050420427273..e94a415ee10a 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -270,7 +270,9 @@ void scx_sub_disable(struct scx_sched *sch)
SCX_CALL_OP(sch, exit, NULL, sch->exit_info);
if (sch->sub_kset)
kobject_del(&sch->sub_kset->kobj);
- kobject_del(&sch->kobj);
+ /* not added if enable failed before scx_sched_sysfs_add() */
+ if (sch->kobj.state_in_sysfs)
+ kobject_del(&sch->kobj);
}

/* verify that a scheduler can be attached to @cgrp and return the parent */
@@ -363,6 +365,10 @@ void scx_sub_enable_workfn(struct kthread_work *work)
if (ret)
goto err_disable;

+ ret = scx_sched_sysfs_add(sch);
+ if (ret)
+ goto err_disable;
+
if (sch->level >= SCX_SUB_MAX_DEPTH) {
scx_error(sch, "max nesting depth %d violated",
SCX_SUB_MAX_DEPTH);
--
2.54.0