Re: [PATCH] sched/fair: avoid creating misfits during cache-aware balancing
From: Tim Chen
Date: Wed Aug 26 2026 - 12:01:49 EST
On Tue, 2026-08-25 at 10:41 -0700, Tim Chen wrote:
> Cache-aware load balancing biases tasks toward their preferred LLC. On
> asymmetric CPU capacity systems (e.g. big.LITTLE) the destination LLC may
> contain CPUs that are too small to run the task. Pulling the task there
> turns it into a misfit, trading a cache-locality gain for a capacity loss
> that's more detrimental to performance.
>
> Guard both cache-aware migration entry points against this:
>
> - can_migrate_llc_task(): forbid the LLC migration when the task fits its
> source CPU but would not fit the destination CPU.
> - alb_break_llc(): veto the active balance under the same condition so the
> runnable task is not pushed onto a CPU that cannot accommodate it.
>
> Both checks are gated with checks for hybrid processors, so symmetric
> systems are unaffected. Tasks that already do not fit their source CPU
> are left to the existing LLC policy, since the move cannot make their
> fitness worse (this also preserves misfit up-migration to bigger CPUs).
>
> Additionally, if there are misfit tasks found in the load balancing
> classification phase, prioritize misfit task migrations
> over LLC load aggregation on asymmetric systems. A better fitting
> CPU will boost performance more than better cache locality.
>
> Reviewed-by: Ricardo Neri <ricardo.neri-calderon@xxxxxxxxxxxxxxx>
> Tested-by: Ricardo Neri <ricardo.neri-calderon@xxxxxxxxxxxxxxx>
> Reviewed-by: Chen Yu <yu.c.chen@xxxxxxxxx>
Forgot my signed off
Signed-off-by: Tim Chen <tim.c.chen@xxxxxxxxxxxxxxx>
Tim
> ---
> kernel/sched/fair.c | 50 ++++++++++++++++++++++++++++++++++++++++-----
> 1 file changed, 45 insertions(+), 5 deletions(-)
>
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index 6d881e530f89..cf5c022bbd55 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -10691,17 +10691,40 @@ static enum llc_mig can_migrate_llc(int src_cpu, int dst_cpu,
> return mig_llc;
> }
>
> +static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct *p)
> +{
> + /*
> + * On asymmetric CPU capacity domains, do not let cache-aware
> + * balancing pull the task onto a destination CPU that cannot
> + * accommodate it. Doing so would turn the task into a misfit on
> + * the destination, trading a cache-locality gain for a capacity
> + * loss. If the task already does not fit its source CPU, the move
> + * cannot make things worse, so let the LLC preference decide.
> + */
> + if ((env->sd->flags & SD_ASYM_CPUCAPACITY) && p &&
> + !task_fits_cpu(p, env->dst_cpu) &&
> + task_fits_cpu(p, env->src_cpu))
> + return true;
> +
> + return false;
> +}
> +
> /*
> * Check if task p can migrate from source LLC to
> * destination LLC in terms of cache aware load balance.
> */
> -static enum llc_mig can_migrate_llc_task(int src_cpu, int dst_cpu,
> +static enum llc_mig can_migrate_llc_task(struct lb_env *env,
> struct task_struct *p)
> {
> struct mm_struct *mm;
> bool to_pref;
> - int cpu;
> + int cpu, src_cpu, dst_cpu;
> +
> + if (task_misfits_asym_cpu(env, p))
> + return mig_forbid;
>
> + src_cpu = env->src_cpu;
> + dst_cpu = env->dst_cpu;
> mm = p->mm;
> if (!mm)
> return mig_unrestricted;
> @@ -10758,6 +10781,14 @@ alb_break_llc(struct lb_env *env)
> unsigned long util = 0;
> struct task_struct *cur;
>
> + /*
> + * Migrating misfit tasks from current CPU
> + * to CPU with a better fit.
> + * Prioritize that over LLC preference.
> + */
> + if (env->migration_type == migrate_misfit)
> + return false;
> +
> if (env->src_rq->nr_running <= 1)
> return true;
>
> @@ -10765,7 +10796,8 @@ alb_break_llc(struct lb_env *env)
> if (cur && cur->sched_class == &fair_sched_class)
> util = task_util(cur);
>
> - if (can_migrate_llc(env->src_cpu, env->dst_cpu,
> + if (task_misfits_asym_cpu(env, cur) ||
> + can_migrate_llc(env->src_cpu, env->dst_cpu,
> util, false) == mig_forbid)
> return true;
> }
> @@ -10805,8 +10837,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
> READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
> return true;
>
> - if (can_migrate_llc_task(env->src_cpu,
> - env->dst_cpu, p) != mig_forbid)
> + if (can_migrate_llc_task(env, p) != mig_forbid)
> return false;
>
> return true;
> @@ -11869,6 +11900,15 @@ static inline bool llc_balance(struct lb_env *env, struct sg_lb_stats *sgs,
> if (env->sd->flags & SD_SHARE_LLC)
> return false;
>
> + /*
> + * On asymmetric domains, group_misfit_task_load
> + * should be prioritized to move tasks to CPU that fit them
> + * over aggregating tasks to their preferred LLC.
> + */
> + if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
> + sgs->group_misfit_task_load)
> + return false;
> +
> /*
> * Skip cache aware tagging if nr_balanced_failed is sufficiently high.
> * Threshold of cache_nice_tries is set to 1 higher than nr_balance_failed