Re: [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases
From: Tim Chen
Date: Tue Oct 06 2026 - 18:25:37 EST
On Fri, 2026-10-02 at 17:44 +0200, Vincent Guittot wrote:
> Update select_task_rq_fair() to be called out of the 3 current cases which
> are :
> - wake up
> - exec
> - fork
>
> We wants to select a rq in some new cases like pushing a runnable task on a
> better CPU than the local one. In such case, it's not a wakeup , nor an
> exec nor a fork. We make sure to not distrub these cases but still
> go through EAS and fast-path.
>
> Signed-off-by: Vincent Guittot <vincent.guittot@xxxxxxxxxx>
> ---
> kernel/sched/core.c | 2 +-
> kernel/sched/fair.c | 57 ++++++++++++++++++++++++++-------------------
> 2 files changed, 34 insertions(+), 25 deletions(-)
>
> diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> index f38cf5a37a8a..837dc74c9a8d 100644
> --- a/kernel/sched/core.c
> +++ b/kernel/sched/core.c
> @@ -3618,7 +3618,7 @@ static int select_fallback_rq(int cpu, struct task_struct *p)
> }
>
> /*
> - * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
> + * The caller (fork, wakeup, push) owns p->pi_lock, ->cpus_ptr is stable.
> */
> static inline
> int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index 19a0e67827f7..f12678850ce2 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -9800,46 +9800,55 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
> }
>
> /*
> - * select_task_rq_fair: Select target runqueue for the waking task in domains
> - * that have the relevant SD flag set. In practice, this is SD_BALANCE_WAKE,
> - * SD_BALANCE_FORK, or SD_BALANCE_EXEC.
> + * select_task_rq_fair: Select a target runqueue for the task.
> + * There are 2 ways to select the target runqueue:
> + * - The fast path which only looks for an idle CPU in the LLC or the smallest
> + * asymmetric domain (i.e. the lowest domain with all compute capacities).
> + * - The slow path which looks for the idlest CPU in the highest domain with
> + * the relevant SD flag set.
> *
> - * Balances load by selecting the idlest CPU in the idlest group, or under
> - * certain conditions an idle sibling CPU if the domain has SD_WAKE_AFFINE set.
> + * In practice, WF_EXEC and WF_FORK uses the slow path whereas WF_TTWU and no
> + * flag (Push) uses the fast path.
> *
> - * Returns the target CPU number.
> */
> static int
> -select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> +select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
> {
> - int sync = (wake_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> + int sync = (select_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> + int want_sibling = !(select_flags & (WF_EXEC | WF_FORK));
> + int new_cpu, cpu = smp_processor_id();
> struct sched_domain *tmp, *sd = NULL;
> - int cpu = smp_processor_id();
> - int new_cpu = prev_cpu;
> - int want_affine = 0;
> /* SD_flags and WF_flags share the first nibble */
> - int sd_flag = wake_flags & 0xF;
> + int sd_flag = select_flags & 0xF;
> + int want_affine = 0;
>
> /*
> - * required for stable ->cpus_allowed
> + * Required for stable ->cpus_allowed
> */
> lockdep_assert_held(&p->pi_lock);
> - if (wake_flags & WF_TTWU) {
> +
> + if (select_flags & WF_TTWU) {
> record_wakee(p);
>
> - if ((wake_flags & WF_CURRENT_CPU) &&
> + if ((select_flags & WF_CURRENT_CPU) &&
> cpumask_test_cpu(cpu, p->cpus_ptr))
> return cpu;
> + }
>
> - if (!is_rd_overutilized(this_rq()->rd)) {
> - new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> - if (new_cpu >= 0)
> - return new_cpu;
> - new_cpu = prev_cpu;
> - }
> + /*
> + * We don't want EAS to be called for exec or fork but it should be
> + * called for any other case such as wake up or push callback.
> + */
> + if (!is_rd_overutilized(this_rq()->rd) && want_sibling) {
> + new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> + if (new_cpu >= 0)
> + return new_cpu;
> + }
>
> + if (select_flags & WF_TTWU)
> want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
> - }
> +
> + new_cpu = prev_cpu;
>
> for_each_domain(cpu, tmp) {
> /*
> @@ -9871,8 +9880,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
>
> /* Fast path */
> - if (wake_flags & WF_TTWU)
> - return select_idle_sibling(p, prev_cpu, new_cpu);
> + if (want_sibling)
> + new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
Hi Vincent,
With this change, a push also goes through select_idle_sibling(), and
select_idle_sibling() sets p->recent_used_cpu = prev on every call.
For a wakeup, prev is the CPU where the task ran last, and the old
value of the hint becomes a second candidate for the next wakeup. For
a push, prev is the CPU where the task is queued now. If the push
doesn't move the task, the hint becomes the current CPU. When the task
later sleeps and wakes up on this CPU, recent_used_cpu is equal to
prev, so the wakeup has no second candidate. Each push attempt that
fails does this again.
Perhaps something like the following fix below on top of the series. It
passes the select flags to select_idle_sibling() and
updates the hint only for WF_TTWU. When fair_push_task() moves the
task, it saves the source CPU as the hint, which is what the hint
holds after a wakeup migration.
Tim
---
kernel/sched/fair.c | 11 +++++++----
1 file changed, 7 insertions(+), 4 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index fd22731949c5..3eb8a0170902 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -1381,7 +1381,7 @@ static bool update_deadline(struct cfs_rq *cfs_rq, struct sched_entity *se)
#include "pelt.h"
-static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu);
+static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu, int select_flags);
static unsigned long task_h_load(struct task_struct *p);
static unsigned long capacity_of(int cpu);
@@ -9072,7 +9072,7 @@ static inline bool asym_fits_cpu(unsigned long util,
/*
* Try and locate an idle core/thread in the LLC cache domain.
*/
-static int select_idle_sibling(struct task_struct *p, int prev, int target)
+static int select_idle_sibling(struct task_struct *p, int prev, int target, int select_flags)
{
bool has_idle_core = false;
struct sched_domain *sd;
@@ -9131,7 +9131,9 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
/* Check a recently used CPU as a potential idle candidate: */
recent_used_cpu = p->recent_used_cpu;
- p->recent_used_cpu = prev;
+ /* A push only updates the hint when it moves the task */
+ if (select_flags & WF_TTWU)
+ p->recent_used_cpu = prev;
if (recent_used_cpu != prev &&
recent_used_cpu != target &&
cpus_share_cache(recent_used_cpu, target) &&
@@ -10015,6 +10017,7 @@ static bool fair_push_task(struct rq *rq)
deactivate_task(rq, next_task, 0);
set_task_cpu(next_task, new_cpu);
+ next_task->recent_used_cpu = prev_cpu;
raw_spin_rq_unlock(rq);
raw_spin_rq_lock(new_rq);
@@ -10225,7 +10228,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
/* Fast path */
if (want_sibling)
- new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
+ new_cpu = select_idle_sibling(p, prev_cpu, new_cpu, select_flags);
return new_cpu;
}