[PATCH v3 1/2] sched/numa: Drive NUMA task tick from execution context

From: Hui Su

Date: Fri Sep 04 2026 - 04:56:39 EST


Proxy execution separates the scheduling context in rq->donor from the
execution context in rq->curr. sched_tick() invokes task_tick() for the
donor's scheduling class.

task_tick_numa() operates on state associated with the task actually
executing, including its mm and NUMA work state. With proxy execution,
rq->donor provides the scheduling context while rq->curr identifies the
execution context.

Task-level execution runtime is likewise accounted to rq->curr, and
task_tick_numa() uses that runtime to drive periodic NUMA scanning.
Keeping task_tick_numa() under task_tick_fair() also means that it is not
invoked when a fair task executes on behalf of an RT or deadline donor.

Move NUMA tick handling into a scheduler helper for the execution
context, and invoke it from both sched_tick() and sched_tick_remote().

Fixes: 7de9d4f94638 ("sched: Start blocked_on chain processing in find_proxy_task()")
Suggested-by: K Prateek Nayak <kprateek.nayak@xxxxxxx>
Suggested-by: Tim Chen <tim.c.chen@xxxxxxxxxxxxxxx>
Signed-off-by: Hui Su <sh_def@xxxxxxx>
---
kernel/sched/core.c | 14 ++++++++++++++
kernel/sched/fair.c | 7 ++-----
kernel/sched/sched.h | 1 +
3 files changed, 17 insertions(+), 5 deletions(-)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index f78275192036..4db55e4ace9e 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5762,6 +5762,17 @@ static int __init setup_resched_latency_warn_ms(char *str)
}
__setup("resched_latency_warn_ms=", setup_resched_latency_warn_ms);

+static void sched_tick_exec_ctx(struct rq *rq)
+{
+ struct task_struct *curr = rq->curr;
+
+ if (curr->sched_class != &fair_sched_class)
+ return;
+
+ if (static_branch_unlikely(&sched_numa_balancing))
+ task_tick_numa(rq, curr);
+}
+
/*
* This function gets called by the timer code, with HZ frequency.
* We call it with interrupts disabled.
@@ -5794,6 +5805,8 @@ void sched_tick(void)
resched_curr(rq);

donor->sched_class->task_tick(rq, donor, 0);
+ sched_tick_exec_ctx(rq);
+
if (sched_feat(LATENCY_WARN))
resched_latency = cpu_resched_latency(rq);
calc_global_load_tick(rq);
@@ -5890,6 +5903,7 @@ static void sched_tick_remote(struct work_struct *work)
WARN_ON_ONCE(delta > (u64)NSEC_PER_SEC * 30);
}
curr->sched_class->task_tick(rq, curr, 0);
+ sched_tick_exec_ctx(rq);

calc_load_nohz_remote(rq);
}
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 8dff37059faf..55f0460e4ae3 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -4425,7 +4425,7 @@ void init_numa_balancing(u64 clone_flags, struct task_struct *p)
/*
* Drive the periodic memory faults..
*/
-static void task_tick_numa(struct rq *rq, struct task_struct *curr)
+void task_tick_numa(struct rq *rq, struct task_struct *curr)
{
struct callback_head *work = &curr->numa_work;
u64 period, now;
@@ -4491,7 +4491,7 @@ static void update_scan_period(struct task_struct *p, int new_cpu)

#else /* !CONFIG_NUMA_BALANCING: */

-static void task_tick_numa(struct rq *rq, struct task_struct *curr)
+void task_tick_numa(struct rq *rq, struct task_struct *curr)
{
}

@@ -15042,9 +15042,6 @@ static void task_tick_fair(struct rq *rq, struct task_struct *curr, int queued)
if (queued)
return;

- if (static_branch_unlikely(&sched_numa_balancing))
- task_tick_numa(rq, curr);
-
task_tick_cache(rq, curr);

update_misfit_status(curr, rq);
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e656c7059bf8..4d619f272b15 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -4152,6 +4152,7 @@ extern void sched_cache_active_set(void);
void sched_domains_free_llc_id(int cpu);

extern void init_sched_mm(struct task_struct *p);
+void task_tick_numa(struct rq *rq, struct task_struct *p);

extern u64 avg_vruntime(struct cfs_rq *cfs_rq);
extern int entity_eligible(struct cfs_rq *cfs_rq, struct sched_entity *se);