Re: [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases
From: Vincent Guittot
Date: Fri Oct 09 2026 - 09:37:49 EST
On Wed, 7 Oct 2026 at 00:25, Tim Chen <tim.c.chen@xxxxxxxxxxxxxxx> wrote:
>
> On Fri, 2026-10-02 at 17:44 +0200, Vincent Guittot wrote:
> > Update select_task_rq_fair() to be called out of the 3 current cases which
> > are :
> > - wake up
> > - exec
> > - fork
> >
> > We wants to select a rq in some new cases like pushing a runnable task on a
> > better CPU than the local one. In such case, it's not a wakeup , nor an
> > exec nor a fork. We make sure to not distrub these cases but still
> > go through EAS and fast-path.
> >
> > Signed-off-by: Vincent Guittot <vincent.guittot@xxxxxxxxxx>
> > ---
> > kernel/sched/core.c | 2 +-
> > kernel/sched/fair.c | 57 ++++++++++++++++++++++++++-------------------
> > 2 files changed, 34 insertions(+), 25 deletions(-)
> >
> > diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> > index f38cf5a37a8a..837dc74c9a8d 100644
> > --- a/kernel/sched/core.c
> > +++ b/kernel/sched/core.c
> > @@ -3618,7 +3618,7 @@ static int select_fallback_rq(int cpu, struct task_struct *p)
> > }
> >
> > /*
> > - * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
> > + * The caller (fork, wakeup, push) owns p->pi_lock, ->cpus_ptr is stable.
> > */
> > static inline
> > int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
> > diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> > index 19a0e67827f7..f12678850ce2 100644
> > --- a/kernel/sched/fair.c
> > +++ b/kernel/sched/fair.c
> > @@ -9800,46 +9800,55 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
> > }
> >
> > /*
> > - * select_task_rq_fair: Select target runqueue for the waking task in domains
> > - * that have the relevant SD flag set. In practice, this is SD_BALANCE_WAKE,
> > - * SD_BALANCE_FORK, or SD_BALANCE_EXEC.
> > + * select_task_rq_fair: Select a target runqueue for the task.
> > + * There are 2 ways to select the target runqueue:
> > + * - The fast path which only looks for an idle CPU in the LLC or the smallest
> > + * asymmetric domain (i.e. the lowest domain with all compute capacities).
> > + * - The slow path which looks for the idlest CPU in the highest domain with
> > + * the relevant SD flag set.
> > *
> > - * Balances load by selecting the idlest CPU in the idlest group, or under
> > - * certain conditions an idle sibling CPU if the domain has SD_WAKE_AFFINE set.
> > + * In practice, WF_EXEC and WF_FORK uses the slow path whereas WF_TTWU and no
> > + * flag (Push) uses the fast path.
> > *
> > - * Returns the target CPU number.
> > */
> > static int
> > -select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> > +select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
> > {
> > - int sync = (wake_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> > + int sync = (select_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> > + int want_sibling = !(select_flags & (WF_EXEC | WF_FORK));
> > + int new_cpu, cpu = smp_processor_id();
> > struct sched_domain *tmp, *sd = NULL;
> > - int cpu = smp_processor_id();
> > - int new_cpu = prev_cpu;
> > - int want_affine = 0;
> > /* SD_flags and WF_flags share the first nibble */
> > - int sd_flag = wake_flags & 0xF;
> > + int sd_flag = select_flags & 0xF;
> > + int want_affine = 0;
> >
> > /*
> > - * required for stable ->cpus_allowed
> > + * Required for stable ->cpus_allowed
> > */
> > lockdep_assert_held(&p->pi_lock);
> > - if (wake_flags & WF_TTWU) {
> > +
> > + if (select_flags & WF_TTWU) {
> > record_wakee(p);
> >
> > - if ((wake_flags & WF_CURRENT_CPU) &&
> > + if ((select_flags & WF_CURRENT_CPU) &&
> > cpumask_test_cpu(cpu, p->cpus_ptr))
> > return cpu;
> > + }
> >
> > - if (!is_rd_overutilized(this_rq()->rd)) {
> > - new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> > - if (new_cpu >= 0)
> > - return new_cpu;
> > - new_cpu = prev_cpu;
> > - }
> > + /*
> > + * We don't want EAS to be called for exec or fork but it should be
> > + * called for any other case such as wake up or push callback.
> > + */
> > + if (!is_rd_overutilized(this_rq()->rd) && want_sibling) {
> > + new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> > + if (new_cpu >= 0)
> > + return new_cpu;
> > + }
> >
> > + if (select_flags & WF_TTWU)
> > want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
> > - }
> > +
> > + new_cpu = prev_cpu;
> >
> > for_each_domain(cpu, tmp) {
> > /*
> > @@ -9871,8 +9880,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> > return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
> >
> > /* Fast path */
> > - if (wake_flags & WF_TTWU)
> > - return select_idle_sibling(p, prev_cpu, new_cpu);
> > + if (want_sibling)
> > + new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
>
> Hi Vincent,
>
> With this change, a push also goes through select_idle_sibling(), and
> select_idle_sibling() sets p->recent_used_cpu = prev on every call.
>
> For a wakeup, prev is the CPU where the task ran last, and the old
> value of the hint becomes a second candidate for the next wakeup. For
> a push, prev is the CPU where the task is queued now. If the push
> doesn't move the task, the hint becomes the current CPU. When the task
> later sleeps and wakes up on this CPU, recent_used_cpu is equal to
> prev, so the wakeup has no second candidate. Each push attempt that
> fails does this again.
I agree that recent_used_cpu will be updated during push selection but
I'm a bit balanced about making it a special case.
When a task wakeup several times on the same prev cpu we have the same
effect for recent_used_cpu.
CPU selection during a push should not differ from CPU selection
during a wakeup. if we end to the same cpu then it seems reasonable
that recent_used_cpu equals prev
>
> Perhaps something like the following fix below on top of the series. It
> passes the select flags to select_idle_sibling() and
> updates the hint only for WF_TTWU. When fair_push_task() moves the
> task, it saves the source CPU as the hint, which is what the hint
> holds after a wakeup migration.
>
> Tim
>
> ---
> kernel/sched/fair.c | 11 +++++++----
> 1 file changed, 7 insertions(+), 4 deletions(-)
>
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index fd22731949c5..3eb8a0170902 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -1381,7 +1381,7 @@ static bool update_deadline(struct cfs_rq *cfs_rq, struct sched_entity *se)
>
> #include "pelt.h"
>
> -static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu);
> +static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu, int select_flags);
> static unsigned long task_h_load(struct task_struct *p);
> static unsigned long capacity_of(int cpu);
>
> @@ -9072,7 +9072,7 @@ static inline bool asym_fits_cpu(unsigned long util,
> /*
> * Try and locate an idle core/thread in the LLC cache domain.
> */
> -static int select_idle_sibling(struct task_struct *p, int prev, int target)
> +static int select_idle_sibling(struct task_struct *p, int prev, int target, int select_flags)
> {
> bool has_idle_core = false;
> struct sched_domain *sd;
> @@ -9131,7 +9131,9 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
>
> /* Check a recently used CPU as a potential idle candidate: */
> recent_used_cpu = p->recent_used_cpu;
> - p->recent_used_cpu = prev;
> + /* A push only updates the hint when it moves the task */
> + if (select_flags & WF_TTWU)
> + p->recent_used_cpu = prev;
> if (recent_used_cpu != prev &&
> recent_used_cpu != target &&
> cpus_share_cache(recent_used_cpu, target) &&
> @@ -10015,6 +10017,7 @@ static bool fair_push_task(struct rq *rq)
>
> deactivate_task(rq, next_task, 0);
> set_task_cpu(next_task, new_cpu);
> + next_task->recent_used_cpu = prev_cpu;
> raw_spin_rq_unlock(rq);
>
> raw_spin_rq_lock(new_rq);
> @@ -10225,7 +10228,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
>
> /* Fast path */
> if (want_sibling)
> - new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
> + new_cpu = select_idle_sibling(p, prev_cpu, new_cpu, select_flags);
>
> return new_cpu;
> }