[PATCH v2 4/6] cpuidle: cache aggregated governor latency QoS constraint
From: Yaxiong Tian
Date: Wed Jul 29 2026 - 02:21:39 EST
cpuidle_governor_latency_req() runs on every idle-state selection and
aggregates per-CPU resume latency with the global CPU latency and
wakeup latency QoS limits. That path repeatedly walks get_cpu_device()
and pm_qos_read_value(), which shows up hot under menu_select().
Cache the aggregated constraint per CPU and refresh it only when the
corresponding per-CPU generation changes. The generation is already
bumped by the global and per-CPU resume latency QoS notifiers added
earlier in this series.
On a menu governor profile, cpuidle_governor_latency_req() drops from
about 19.9% of menu_select time to about 4.2%, roughly a 6x reduction
in cost per call (~1.9 us down to ~0.3 us).
Signed-off-by: Yaxiong Tian <tianyaxiong@xxxxxxxxxx>
---
drivers/cpuidle/governor.c | 46 +++++++++++++++++++++++++++++++++++---
1 file changed, 43 insertions(+), 3 deletions(-)
diff --git a/drivers/cpuidle/governor.c b/drivers/cpuidle/governor.c
index d286ccf19a69..f0be7c0cc612 100644
--- a/drivers/cpuidle/governor.c
+++ b/drivers/cpuidle/governor.c
@@ -29,14 +29,23 @@ struct cpuidle_governor *cpuidle_prev_governor;
* Per-CPU generation bumped to invalidate that CPU's cached latency
* constraint. Global QoS changes invalidate every CPU; per-CPU resume
* latency changes invalidate only the affected CPU.
+ *
+ * Start at 1 so it does not match a zero-filled latency_req_cache and
+ * falsely hit before the first real fill (which would return 0).
*/
-static DEFINE_PER_CPU(atomic_t, latency_req_gen);
+static DEFINE_PER_CPU(atomic_t, latency_req_gen) = ATOMIC_INIT(1);
+
+struct cpuidle_latency_req_cache {
+ unsigned int gen;
+ s64 latency_ns;
+};
struct cpuidle_cpu_qos_nb {
struct notifier_block nb;
unsigned int cpu;
};
+static DEFINE_PER_CPU(struct cpuidle_latency_req_cache, latency_req_cache);
static DEFINE_PER_CPU(struct cpuidle_cpu_qos_nb, cpuidle_cpu_qos_nb);
static void cpuidle_latency_req_invalidate_cpu(unsigned int cpu)
@@ -207,10 +216,14 @@ int cpuidle_register_governor(struct cpuidle_governor *gov)
}
/**
- * cpuidle_governor_latency_req - Compute a latency constraint for CPU
+ * cpuidle_aggregate_latency_req - Aggregate QoS latency constraints for @cpu
* @cpu: Target CPU
+ *
+ * Combine the per-CPU resume latency with the global CPU latency and wakeup
+ * latency QoS limits. Called on a cache miss from
+ * cpuidle_governor_latency_req().
*/
-s64 cpuidle_governor_latency_req(unsigned int cpu)
+static s64 cpuidle_aggregate_latency_req(unsigned int cpu)
{
struct device *device = get_cpu_device(cpu);
int device_req = dev_pm_qos_raw_resume_latency(device);
@@ -225,3 +238,30 @@ s64 cpuidle_governor_latency_req(unsigned int cpu)
return (s64)device_req * NSEC_PER_USEC;
}
+
+/**
+ * cpuidle_governor_latency_req - Cached latency constraint for CPU
+ * @cpu: Target CPU (must be the local CPU; callers are idle governors)
+ *
+ * The per-CPU cache is only accessed by this CPU's idle ->select() path.
+ * Remote CPUs may bump latency_req_gen via QoS notifiers, but they do not
+ * touch the cache fields.
+ */
+s64 cpuidle_governor_latency_req(unsigned int cpu)
+{
+ struct cpuidle_latency_req_cache *cache;
+ unsigned int gen;
+ s64 latency_ns;
+
+ cache = per_cpu_ptr(&latency_req_cache, cpu);
+ gen = atomic_read(per_cpu_ptr(&latency_req_gen, cpu));
+
+ if (likely(cache->gen == gen))
+ return cache->latency_ns;
+
+ latency_ns = cpuidle_aggregate_latency_req(cpu);
+ cache->latency_ns = latency_ns;
+ cache->gen = gen;
+
+ return latency_ns;
+}
--
2.43.0