[PATCH v2 4/6] cpuidle: cache aggregated governor latency QoS constraint

Yaxiong Tian <[email protected]>
Newsgroups org.kernel.vger.linux-pm,org.kernel.vger.linux-kernel,org.kernel.vger.linux-kselftest
Message-ID <[email protected]>
cpuidle_governor_latency_req() runs on every idle-state selection and
aggregates per-CPU resume latency with the global CPU latency and
wakeup latency QoS limits.  That path repeatedly walks get_cpu_device()
and pm_qos_read_value(), which shows up hot under menu_select().

Cache the aggregated constraint per CPU and refresh it only when the
corresponding per-CPU generation changes.  The generation is already
bumped by the global and per-CPU resume latency QoS notifiers added
earlier in this series.

On a menu governor profile, cpuidle_governor_latency_req() drops from
about 19.9% of menu_select time to about 4.2%, roughly a 6x reduction
in cost per call (~1.9 us down to ~0.3 us).

Signed-off-by: Yaxiong Tian <[email protected]>
---
 drivers/cpuidle/governor.c | 46 +++++++++++++++++++++++++++++++++++---
 1 file changed, 43 insertions(+), 3 deletions(-)

diff --git a/drivers/cpuidle/governor.c b/drivers/cpuidle/governor.c
index d286ccf19a69..f0be7c0cc612 100644
--- a/drivers/cpuidle/governor.c
+++ b/drivers/cpuidle/governor.c
@@ -29,14 +29,23 @@ struct cpuidle_governor *cpuidle_prev_governor;
  * Per-CPU generation bumped to invalidate that CPU's cached latency
  * constraint.  Global QoS changes invalidate every CPU; per-CPU resume
  * latency changes invalidate only the affected CPU.
+ *
+ * Start at 1 so it does not match a zero-filled latency_req_cache and
+ * falsely hit before the first real fill (which would return 0).
  */
-static DEFINE_PER_CPU(atomic_t, latency_req_gen);
+static DEFINE_PER_CPU(atomic_t, latency_req_gen) = ATOMIC_INIT(1);
+
+struct cpuidle_latency_req_cache {
+	unsigned int gen;
+	s64 latency_ns;
+};
 
 struct cpuidle_cpu_qos_nb {
 	struct notifier_block nb;
 	unsigned int cpu;
 };
 
+static DEFINE_PER_CPU(struct cpuidle_latency_req_cache, latency_req_cache);
 static DEFINE_PER_CPU(struct cpuidle_cpu_qos_nb, cpuidle_cpu_qos_nb);
 
 static void cpuidle_latency_req_invalidate_cpu(unsigned int cpu)
@@ -207,10 +216,14 @@ int cpuidle_register_governor(struct cpuidle_governor *gov)
 }
 
 /**
- * cpuidle_governor_latency_req - Compute a latency constraint for CPU
+ * cpuidle_aggregate_latency_req - Aggregate QoS latency constraints for @cpu
  * @cpu: Target CPU
+ *
+ * Combine the per-CPU resume latency with the global CPU latency and wakeup
+ * latency QoS limits.  Called on a cache miss from
+ * cpuidle_governor_latency_req().
  */
-s64 cpuidle_governor_latency_req(unsigned int cpu)
+static s64 cpuidle_aggregate_latency_req(unsigned int cpu)
 {
 	struct device *device = get_cpu_device(cpu);
 	int device_req = dev_pm_qos_raw_resume_latency(device);
@@ -225,3 +238,30 @@ s64 cpuidle_governor_latency_req(unsigned int cpu)
 
 	return (s64)device_req * NSEC_PER_USEC;
 }
+
+/**
+ * cpuidle_governor_latency_req - Cached latency constraint for CPU
+ * @cpu: Target CPU (must be the local CPU; callers are idle governors)
+ *
+ * The per-CPU cache is only accessed by this CPU's idle ->select() path.
+ * Remote CPUs may bump latency_req_gen via QoS notifiers, but they do not
+ * touch the cache fields.
+ */
+s64 cpuidle_governor_latency_req(unsigned int cpu)
+{
+	struct cpuidle_latency_req_cache *cache;
+	unsigned int gen;
+	s64 latency_ns;
+
+	cache = per_cpu_ptr(&latency_req_cache, cpu);
+	gen = atomic_read(per_cpu_ptr(&latency_req_gen, cpu));
+
+	if (likely(cache->gen == gen))
+		return cache->latency_ns;
+
+	latency_ns = cpuidle_aggregate_latency_req(cpu);
+	cache->latency_ns = latency_ns;
+	cache->gen = gen;
+
+	return latency_ns;
+}
-- 
2.43.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.