[PATCH 3/5] sched_ext: Scan NUMA hinting faults for opted-in BPF schedulers

From: Andrea Righi

Date: Sun Oct 04 2026 - 03:30:11 EST


A task running under a BPF scheduler never has its address space scanned
for NUMA hinting faults. As a result, mm/memory.c never calls
task_numa_fault() and the task's NUMA fault statistics and preferred
node remain empty.

Scanning is not free and a scheduler which does not use the result
should not pay for it. Add SCX_OPS_NUMA_BALANCING to let a BPF scheduler
request the NUMA hinting-fault scan for its tasks. Drive the scan from
the sched_ext tick for tasks owned by an opted-in scheduler, the way
task_tick_fair() does, under the existing sched_numa_balancing static
key. The hrtick, which only fires to end the running task's slice, does
not drive the scan, as in task_tick_fair().

The hinting faults then work for these tasks as they do for fair tasks:
they maintain the NUMA statistics, the preferred node and potentially
migrate the memory a task accesses toward the node the task runs on. The
tasks themselves are not migrated, the BPF scheduler remains responsible
for placing them.

No new accounting is needed to pace the scan, task_tick_numa() spaces
scans by the task's CPU time, p->se.sum_exec_runtime and sched_ext
already maintains that field through update_curr_common().

Reported-by: Vladimir Vdovin <deliran@xxxxxxxxxx>
Link: https://lore.kernel.org/r/20261002124559.10367-1-deliran@xxxxxxxxxx
Signed-off-by: Andrea Righi <arighi@xxxxxxxxxx>
---
kernel/sched/ext/ext.c | 13 +++++++++++++
kernel/sched/ext/internal.h | 16 +++++++++++++++-
tools/sched_ext/include/scx/compat.h | 1 +
tools/sched_ext/include/scx/enum_defs.autogen.h | 1 +
tools/sched_ext/include/scx/enums_abi.autogen.h | 3 ++-
5 files changed, 32 insertions(+), 2 deletions(-)

diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 96c904b396019..b22db5ccb91ab 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -4227,6 +4227,19 @@ static void task_tick_scx(struct rq *rq, struct task_struct *donor, int queued)
else if (SCX_HAS_OP(sch, tick))
SCX_CALL_OP_TASK(sch, tick, rq, donor);

+ /*
+ * If requested by the scheduler, drive the NUMA hinting-fault scan as
+ * task_tick_fair() does. Neither the scan nor the statistics it feeds
+ * depend on the scheduling class. Where the task then runs is left to
+ * that scheduler: numa_migrate_preferred() does not move a task it owns.
+ *
+ * The tick that only refreshes an already queued task does not scan,
+ * matching the @queued check in task_tick_fair().
+ */
+ if (!queued && (sch->ops.flags & SCX_OPS_NUMA_BALANCING) &&
+ static_branch_unlikely(&sched_numa_balancing))
+ task_tick_numa(rq, donor);
+
if (!donor->scx.slice) {
/* the slice can't be trusted while bypassing */
if (READ_ONCE(donor->scx.lazy_resched) &&
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index d65fec631bdf7..10cbbee48e094 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -243,6 +243,19 @@ enum scx_ops_flags {
*/
SCX_OPS_ENQ_BLOCKED = 1LLU << 10,

+ /*
+ * If set, drive automatic NUMA hinting-fault scans for tasks owned by
+ * this scheduler. The faults maintain the tasks' NUMA statistics and
+ * preferred node, which can be queried with scx_bpf_task_numa_nid(),
+ * and migrate the memory a task accesses toward the node it runs on, as
+ * NUMA balancing does for fair tasks. Tasks themselves are never
+ * migrated: the BPF scheduler remains responsible for placing them.
+ *
+ * If clear, sched_ext does not initiate NUMA hinting-fault scans for the
+ * scheduler's tasks.
+ */
+ SCX_OPS_NUMA_BALANCING = 1LLU << 11,
+
SCX_OPS_ALL_FLAGS = SCX_OPS_KEEP_BUILTIN_IDLE |
SCX_OPS_ENQ_LAST |
SCX_OPS_ENQ_EXITING |
@@ -253,7 +266,8 @@ enum scx_ops_flags {
SCX_OPS_ALWAYS_ENQ_IMMED |
SCX_OPS_TID_TO_TASK |
SCX_OPS_LAZY_RESCHED |
- SCX_OPS_ENQ_BLOCKED,
+ SCX_OPS_ENQ_BLOCKED |
+ SCX_OPS_NUMA_BALANCING,

/* high 8 bits are internal, don't include in SCX_OPS_ALL_FLAGS */
__SCX_OPS_INTERNAL_MASK = 0xffLLU << 56,
diff --git a/tools/sched_ext/include/scx/compat.h b/tools/sched_ext/include/scx/compat.h
index 07fb55e63c57c..f4c35f8750730 100644
--- a/tools/sched_ext/include/scx/compat.h
+++ b/tools/sched_ext/include/scx/compat.h
@@ -215,6 +215,7 @@ static inline bool __COMPAT_struct_has_field(const char *type, const char *field
#define SCX_OPS_BUILTIN_IDLE_PER_NODE SCX_OPS_FLAG(SCX_OPS_BUILTIN_IDLE_PER_NODE)
#define SCX_OPS_ALWAYS_ENQ_IMMED SCX_OPS_FLAG(SCX_OPS_ALWAYS_ENQ_IMMED)
#define SCX_OPS_ENQ_BLOCKED SCX_OPS_FLAG(SCX_OPS_ENQ_BLOCKED)
+#define SCX_OPS_NUMA_BALANCING SCX_OPS_FLAG(SCX_OPS_NUMA_BALANCING)

#define SCX_PICK_IDLE_FLAG(name) __COMPAT_ENUM_OR_ZERO("scx_pick_idle_cpu_flags", #name)

diff --git a/tools/sched_ext/include/scx/enum_defs.autogen.h b/tools/sched_ext/include/scx/enum_defs.autogen.h
index 4fa82033b99c9..1bed9c9a96f29 100644
--- a/tools/sched_ext/include/scx/enum_defs.autogen.h
+++ b/tools/sched_ext/include/scx/enum_defs.autogen.h
@@ -172,6 +172,7 @@
#define HAVE_SCX_OPS_TID_TO_TASK
#define HAVE_SCX_OPS_LAZY_RESCHED
#define HAVE_SCX_OPS_ENQ_BLOCKED
+#define HAVE_SCX_OPS_NUMA_BALANCING
#define HAVE_SCX_OPS_ALL_FLAGS
#define HAVE___SCX_OPS_INTERNAL_MASK
#define HAVE_SCX_OPS_HAS_CPU_PREEMPT
diff --git a/tools/sched_ext/include/scx/enums_abi.autogen.h b/tools/sched_ext/include/scx/enums_abi.autogen.h
index 1186daa9cb141..1cc1aad845023 100644
--- a/tools/sched_ext/include/scx/enums_abi.autogen.h
+++ b/tools/sched_ext/include/scx/enums_abi.autogen.h
@@ -184,7 +184,8 @@ static const struct __scx_enum_abi_val __scx_enum_abi_vals[]
{ "scx_ops_flags", "SCX_OPS_TID_TO_TASK", 0x100LLU },
{ "scx_ops_flags", "SCX_OPS_LAZY_RESCHED", 0x200LLU },
{ "scx_ops_flags", "SCX_OPS_ENQ_BLOCKED", 0x400LLU },
- { "scx_ops_flags", "SCX_OPS_ALL_FLAGS", 0x7ffLLU },
+ { "scx_ops_flags", "SCX_OPS_NUMA_BALANCING", 0x800LLU },
+ { "scx_ops_flags", "SCX_OPS_ALL_FLAGS", 0xfffLLU },
{ "scx_ops_flags", "__SCX_OPS_INTERNAL_MASK", 0xff00000000000000LLU },
{ "scx_ops_flags", "SCX_OPS_HAS_CPU_PREEMPT", 0x100000000000000LLU },
{ "scx_ops_state", "SCX_OPSS_NONE", 0x0LLU },
--
2.55.0