[PATCH 2/3] powerpc/perf: Use the aggregate context switch values from vcpu struct

From: Gautam Menghani

Date: Tue Aug 11 2026 - 10:12:23 EST


The vpa-pmu driver reports incorrect numbers in 2 scenarios:

1. The vCPU process gets rescheduled to a different host cpu
- Incorrect numbers are observed here because the PACA is per-host cpu
resource, and the KVM vCPUs can be rescheduled to different
host cpus. This causes the vpa-pmu driver to subtract wrong values
when vCPUs are rescheduled.

2. The vCPU is not running when vpa_pmu_read() is called.
- In this case get_counter_data() returns 0, and this can result in
negative numbers getting reported.

Fix the above issues by using the aggregate values from the vcpu
structure to capture and report the difference in counter values.

Signed-off-by: Gautam Menghani <gautam@xxxxxxxxxxxxx>
---
arch/powerpc/include/asm/kvm_book3s_64.h | 6 ---
arch/powerpc/kvm/book3s_hv.c | 63 ----------------------
arch/powerpc/perf/vpa-pmu.c | 69 ++++++++++++------------
3 files changed, 35 insertions(+), 103 deletions(-)

diff --git a/arch/powerpc/include/asm/kvm_book3s_64.h b/arch/powerpc/include/asm/kvm_book3s_64.h
index b936e174eefd..11065313d4c1 100644
--- a/arch/powerpc/include/asm/kvm_book3s_64.h
+++ b/arch/powerpc/include/asm/kvm_book3s_64.h
@@ -688,12 +688,6 @@ int kvmhv_counters_tracepoint_regfunc(void);
void kvmhv_counters_tracepoint_unregfunc(void);
int kvmhv_get_l2_counters_status(void);
void kvmhv_set_l2_counters_status(int cpu, bool status);
-u64 kvmhv_get_l1_to_l2_cs_time(void);
-u64 kvmhv_get_l2_to_l1_cs_time(void);
-u64 kvmhv_get_l2_runtime_agg(void);
-u64 kvmhv_get_l1_to_l2_cs_time_vcpu(void);
-u64 kvmhv_get_l2_to_l1_cs_time_vcpu(void);
-u64 kvmhv_get_l2_runtime_agg_vcpu(void);

#endif /* CONFIG_KVM_BOOK3S_HV_POSSIBLE */

diff --git a/arch/powerpc/kvm/book3s_hv.c b/arch/powerpc/kvm/book3s_hv.c
index 0e8a959122a8..b315a959d2a1 100644
--- a/arch/powerpc/kvm/book3s_hv.c
+++ b/arch/powerpc/kvm/book3s_hv.c
@@ -4172,69 +4172,6 @@ static void do_trace_nested_cs_time(struct kvm_vcpu *vcpu)
*l2_runtime_agg_ptr = l2_runtime_ns;
}

-u64 kvmhv_get_l1_to_l2_cs_time(void)
-{
- return tb_to_ns(be64_to_cpu(get_lppaca()->l1_to_l2_cs_tb));
-}
-EXPORT_SYMBOL(kvmhv_get_l1_to_l2_cs_time);
-
-u64 kvmhv_get_l2_to_l1_cs_time(void)
-{
- return tb_to_ns(be64_to_cpu(get_lppaca()->l2_to_l1_cs_tb));
-}
-EXPORT_SYMBOL(kvmhv_get_l2_to_l1_cs_time);
-
-u64 kvmhv_get_l2_runtime_agg(void)
-{
- return tb_to_ns(be64_to_cpu(get_lppaca()->l2_runtime_tb));
-}
-EXPORT_SYMBOL(kvmhv_get_l2_runtime_agg);
-
-u64 kvmhv_get_l1_to_l2_cs_time_vcpu(void)
-{
- struct kvm_vcpu *vcpu;
- struct kvm_vcpu_arch *arch;
-
- vcpu = local_paca->kvm_hstate.kvm_vcpu;
- if (vcpu) {
- arch = &vcpu->arch;
- return arch->l1_to_l2_cs;
- } else {
- return 0;
- }
-}
-EXPORT_SYMBOL(kvmhv_get_l1_to_l2_cs_time_vcpu);
-
-u64 kvmhv_get_l2_to_l1_cs_time_vcpu(void)
-{
- struct kvm_vcpu *vcpu;
- struct kvm_vcpu_arch *arch;
-
- vcpu = local_paca->kvm_hstate.kvm_vcpu;
- if (vcpu) {
- arch = &vcpu->arch;
- return arch->l2_to_l1_cs;
- } else {
- return 0;
- }
-}
-EXPORT_SYMBOL(kvmhv_get_l2_to_l1_cs_time_vcpu);
-
-u64 kvmhv_get_l2_runtime_agg_vcpu(void)
-{
- struct kvm_vcpu *vcpu;
- struct kvm_vcpu_arch *arch;
-
- vcpu = local_paca->kvm_hstate.kvm_vcpu;
- if (vcpu) {
- arch = &vcpu->arch;
- return arch->l2_runtime_agg;
- } else {
- return 0;
- }
-}
-EXPORT_SYMBOL(kvmhv_get_l2_runtime_agg_vcpu);
-
#else
int kvmhv_get_l2_counters_status(void)
{
diff --git a/arch/powerpc/perf/vpa-pmu.c b/arch/powerpc/perf/vpa-pmu.c
index bff4cfab7b94..334ef7719db9 100644
--- a/arch/powerpc/perf/vpa-pmu.c
+++ b/arch/powerpc/perf/vpa-pmu.c
@@ -71,6 +71,28 @@ static const struct attribute_group *vpa_pmu_attr_groups[] = {
NULL
};

+static u64 get_vcpu_data(struct kvm_vcpu *vcpu, u64 config)
+{
+ u64 new_data;
+
+ if (!vcpu)
+ return 0;
+
+ switch (config) {
+ case L1_TO_L2_CS_LAT:
+ new_data = vcpu->arch.l1_to_l2_cs;
+ break;
+ case L2_TO_L1_CS_LAT:
+ new_data = vcpu->arch.l2_to_l1_cs;
+ break;
+ case L2_RUNTIME_AGG:
+ new_data = vcpu->arch.l2_runtime_agg;
+ break;
+ }
+
+ return new_data;
+}
+
static int vpa_pmu_event_init(struct perf_event *event)
{
if (event->attr.type != event->pmu->type)
@@ -91,56 +113,35 @@ static int vpa_pmu_event_init(struct perf_event *event)
return 0;
}

-static unsigned long get_counter_data(struct perf_event *event)
-{
- unsigned int config = event->attr.config;
- u64 data;
-
- switch (config) {
- case L1_TO_L2_CS_LAT:
- if (event->attach_state & PERF_ATTACH_TASK)
- data = kvmhv_get_l1_to_l2_cs_time_vcpu();
- else
- data = kvmhv_get_l1_to_l2_cs_time();
- break;
- case L2_TO_L1_CS_LAT:
- if (event->attach_state & PERF_ATTACH_TASK)
- data = kvmhv_get_l2_to_l1_cs_time_vcpu();
- else
- data = kvmhv_get_l2_to_l1_cs_time();
- break;
- case L2_RUNTIME_AGG:
- if (event->attach_state & PERF_ATTACH_TASK)
- data = kvmhv_get_l2_runtime_agg_vcpu();
- else
- data = kvmhv_get_l2_runtime_agg();
- break;
- default:
- data = 0;
- break;
- }
-
- return data;
-}
-
static int vpa_pmu_add(struct perf_event *event, int flags)
{
u64 data;
+ struct kvm_vcpu *vcpu;
+
+ vcpu = local_paca->kvm_hstate.kvm_vcpu;
+ if (!vcpu)
+ goto out;

+ event->pmu_private = vcpu;
kvmhv_set_l2_counters_status(smp_processor_id(), true);

- data = get_counter_data(event);
+ data = get_vcpu_data(vcpu, event->attr.config);
local64_set(&event->hw.prev_count, data);

+out:
return 0;
}

static void vpa_pmu_read(struct perf_event *event)
{
u64 prev_data, new_data, final_data;
+ struct kvm_vcpu *vcpu;

+ vcpu = (struct kvm_vcpu *) event->pmu_private;
+ if (!vcpu)
+ return;
prev_data = local64_read(&event->hw.prev_count);
- new_data = get_counter_data(event);
+ new_data = get_vcpu_data(vcpu, event->attr.config);
final_data = new_data - prev_data;

local64_add(final_data, &event->count);
--
2.54.0