[PATCH v9 12/22] KVM: arm64: Context swap Partitioned PMU guest registers

From: Colton Lewis

Date: Thu Sep 24 2026 - 13:46:26 EST


Save and restore newly untrapped registers that can be directly
accessed by the guest when the PMU is partitioned.

- PMEVCNTRn_EL0
- PMCCNTR_EL0
- PMSELR_EL0
- PMCR_EL0
- PMCNTEN_EL0
- PMINTEN_EL1

If we know we are not partitioned (that is, using the emulated vPMU),
then return immediately. A later patch will make this lazy so the
context swaps don't happen unless the guest has accessed the PMU.

PMEVTYPER is handled in a following patch since we must apply the KVM
event filter before writing values to hardware.

PMOVS guest counters are cleared to avoid the possibility of
generating spurious interrupts when PMINTEN is written. This is fine
because the virtual register for PMOVS is always the canonical value.

Signed-off-by: Colton Lewis <coltonlewis@xxxxxxxxxx>
---
arch/arm64/kvm/arm.c | 4 +-
arch/arm64/kvm/pmu-direct.c | 153 +++++++++++++++++++++++++++++++++++-
arch/arm64/kvm/sys_regs.c | 6 +-
include/kvm/arm_pmu.h | 5 ++
4 files changed, 164 insertions(+), 4 deletions(-)

diff --git a/arch/arm64/kvm/arm.c b/arch/arm64/kvm/arm.c
index 8b080804bc90b..75e0f746623ec 100644
--- a/arch/arm64/kvm/arm.c
+++ b/arch/arm64/kvm/arm.c
@@ -714,6 +714,7 @@ void kvm_arch_vcpu_load(struct kvm_vcpu *vcpu, int cpu)
if (has_vhe())
kvm_vcpu_load_vhe(vcpu);
kvm_arch_vcpu_load_fp(vcpu);
+ kvm_pmu_load(vcpu);
kvm_vcpu_pmu_restore_guest(vcpu);
if (kvm_arm_is_pvtime_enabled(&vcpu->arch))
kvm_make_request(KVM_REQ_RECORD_STEAL, vcpu);
@@ -755,13 +756,14 @@ void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu)
vcpu_set_flag(vcpu, PKVM_HOST_STATE_DIRTY);
}

+ kvm_pmu_put(vcpu);
+ kvm_vcpu_pmu_restore_host(vcpu);
kvm_vcpu_put_debug(vcpu);
kvm_arch_vcpu_put_fp(vcpu);
if (has_vhe())
kvm_vcpu_put_vhe(vcpu);
kvm_timer_vcpu_put(vcpu);
kvm_vgic_put(vcpu);
- kvm_vcpu_pmu_restore_host(vcpu);
if (vcpu_has_nv(vcpu))
kvm_vcpu_put_hw_mmu(vcpu);
kvm_arm_vmid_clear_active();
diff --git a/arch/arm64/kvm/pmu-direct.c b/arch/arm64/kvm/pmu-direct.c
index 5045051e91cd3..92eec0fa33867 100644
--- a/arch/arm64/kvm/pmu-direct.c
+++ b/arch/arm64/kvm/pmu-direct.c
@@ -155,7 +155,6 @@ static u64 kvm_vcpu_pmu_guest_counter_mask(struct kvm_vcpu *vcpu)

return 0;
}
-
/**
* kvm_pmu_guest_counter_mask() - Compute bitmask of guest-reserved counters
*
@@ -169,3 +168,155 @@ u64 kvm_pmu_guest_counter_mask(void)
{
return kvm_vcpu_pmu_guest_counter_mask(kvm_get_running_vcpu());
}
+
+/**
+ * kvm_pmu_load() - Load untrapped PMU registers
+ * @vcpu: Pointer to struct kvm_vcpu
+ *
+ * Load all untrapped PMU registers from the VCPU into the PCPU. Mask
+ * to only bits belonging to guest-reserved counters and leave
+ * host-reserved counters alone in bitmask registers.
+ */
+void kvm_pmu_load(struct kvm_vcpu *vcpu)
+{
+ unsigned long guest_counters;
+ u64 mask;
+ u8 i;
+ u64 val;
+
+ /*
+ * If we aren't guest-owned then we know the guest isn't using
+ * the PMU anyway, so no need to bother with the swap.
+ */
+ if (!kvm_pmu_is_partitioned(vcpu->kvm))
+ return;
+
+ preempt_disable();
+
+ guest_counters = kvm_vcpu_pmu_guest_counter_mask(vcpu);
+
+ for_each_set_bit(i, &guest_counters, ARMPMU_MAX_HWEVENTS) {
+ val = __vcpu_sys_reg(vcpu, PMEVCNTR0_EL0 + i);
+
+ if (i == ARMV8_PMU_CYCLE_IDX)
+ write_pmccntr(val);
+ else
+ write_pmevcntrn(i, val);
+ }
+
+ val = __vcpu_sys_reg(vcpu, PMSELR_EL0);
+ write_sysreg(val, pmselr_el0);
+
+ if (!(vcpu->arch.mdcr_el2 & MDCR_EL2_TPM)) {
+ val = __vcpu_sys_reg(vcpu, PMUSERENR_EL0);
+ write_sysreg(val, pmuserenr_el0);
+ }
+
+ /* Save only the stateful writable bits. */
+ val = __vcpu_sys_reg(vcpu, PMCR_EL0);
+ mask = ARMV8_PMU_PMCR_MASK &
+ ~(ARMV8_PMU_PMCR_P | ARMV8_PMU_PMCR_C);
+ write_sysreg(val & mask, pmcr_el0);
+
+ /*
+ * When handling these:
+ * 1. Apply only the bits for guest counters (indicated by mask)
+ * 2. Use the different registers for set and clear
+ */
+ mask = guest_counters;
+
+ /* Clear the hardware overflow flags so there is no chance of
+ * creating spurious interrupts. The hardware here is never
+ * the canonical version anyway.
+ */
+ write_sysreg(mask, pmovsclr_el0);
+
+ val = __vcpu_sys_reg(vcpu, PMCNTENSET_EL0);
+ write_sysreg(val & mask, pmcntenset_el0);
+ write_sysreg(~val & mask, pmcntenclr_el0);
+
+ val = __vcpu_sys_reg(vcpu, PMINTENSET_EL1);
+ write_sysreg(val & mask, pmintenset_el1);
+ write_sysreg(~val & mask, pmintenclr_el1);
+
+ preempt_enable();
+}
+
+/**
+ * kvm_pmu_put() - Put untrapped PMU registers
+ * @vcpu: Pointer to struct kvm_vcpu
+ *
+ * Put all untrapped PMU registers from the VCPU into the PCPU. Mask
+ * to only bits belonging to guest-reserved counters and leave
+ * host-reserved counters alone in bitmask registers.
+ */
+void kvm_pmu_put(struct kvm_vcpu *vcpu)
+{
+ unsigned long guest_counters;
+ unsigned long flags;
+ u64 mask;
+ u8 i;
+ u64 val;
+
+ /*
+ * If we aren't guest-owned then we know the guest is not
+ * accessing the PMU anyway, so no need to bother with the
+ * swap.
+ */
+ if (!kvm_pmu_is_partitioned(vcpu->kvm))
+ return;
+
+ preempt_disable();
+
+ guest_counters = kvm_vcpu_pmu_guest_counter_mask(vcpu);
+ mask = guest_counters;
+
+ /* Mask these to only save the guest relevant bits. */
+ val = read_sysreg(pmcntenset_el0);
+ __vcpu_assign_sys_reg(vcpu, PMCNTENSET_EL0, val & mask);
+
+ val = read_sysreg(pmintenset_el1);
+ __vcpu_assign_sys_reg(vcpu, PMINTENSET_EL1, val & mask);
+
+ /* Stop guest counters and disable interrupts in hardware first. */
+ write_sysreg(mask, pmcntenclr_el0);
+ write_sysreg(mask, pmintenclr_el1);
+ isb();
+
+ for_each_set_bit(i, &guest_counters, ARMPMU_MAX_HWEVENTS) {
+ if (i == ARMV8_PMU_CYCLE_IDX)
+ val = read_pmccntr();
+ else
+ val = read_pmevcntrn(i);
+
+ __vcpu_assign_sys_reg(vcpu, PMEVCNTR0_EL0 + i, val);
+ }
+
+ val = read_sysreg(pmselr_el0);
+ __vcpu_assign_sys_reg(vcpu, PMSELR_EL0, val);
+
+ if (!(vcpu->arch.mdcr_el2 & MDCR_EL2_TPM)) {
+ val = read_sysreg(pmuserenr_el0);
+ __vcpu_assign_sys_reg(vcpu, PMUSERENR_EL0, val);
+ }
+
+ val = read_sysreg(pmcr_el0);
+ __vcpu_rmw_sys_reg(vcpu, PMCR_EL0, &=, ~ARMV8_PMU_PMCR_MASK);
+ __vcpu_rmw_sys_reg(vcpu, PMCR_EL0, |=, val & ARMV8_PMU_PMCR_MASK);
+
+ val = ARMV8_PMU_PMCR_LC;
+ if (pmu && pmu->pmuver >= ID_AA64DFR0_EL1_PMUVer_V3P5)
+ val |= ARMV8_PMU_PMCR_LP;
+ if (vcpu->arch.mdcr_el2 & MDCR_EL2_HPME)
+ val |= ARMV8_PMU_PMCR_E;
+ write_sysreg(val, pmcr_el0);
+
+ /* Save pending guest hardware overflows. */
+ local_irq_save(flags);
+ val = read_sysreg(pmovsset_el0);
+ __vcpu_rmw_sys_reg(vcpu, PMOVSSET_EL0, |=, val & mask);
+ write_sysreg(val & mask, pmovsclr_el0);
+ local_irq_restore(flags);
+
+ preempt_enable();
+}
diff --git a/arch/arm64/kvm/sys_regs.c b/arch/arm64/kvm/sys_regs.c
index c8210a8b10389..ebcf52261df65 100644
--- a/arch/arm64/kvm/sys_regs.c
+++ b/arch/arm64/kvm/sys_regs.c
@@ -1189,7 +1189,8 @@ static void pmu_reg_write(struct kvm_vcpu *vcpu, enum vcpu_sysreg reg, u64 val,
local_irq_restore(flags);
break;
case PMUSERENR_EL0:
- if (kvm_pmu_is_partitioned(vcpu->kvm))
+ if (kvm_pmu_is_partitioned(vcpu->kvm) &&
+ !(vcpu->arch.mdcr_el2 & MDCR_EL2_TPM))
write_sysreg(val, pmuserenr_el0);
__vcpu_assign_sys_reg(vcpu, reg, val);
break;
@@ -1274,7 +1275,8 @@ static u64 pmu_reg_read(struct kvm_vcpu *vcpu, enum vcpu_sysreg reg)
local_irq_restore(flags);
break;
case PMUSERENR_EL0:
- if (kvm_pmu_is_partitioned(vcpu->kvm))
+ if (kvm_pmu_is_partitioned(vcpu->kvm) &&
+ !(vcpu->arch.mdcr_el2 & MDCR_EL2_TPM))
val = read_sysreg(pmuserenr_el0);
else
val = __vcpu_sys_reg(vcpu, reg);
diff --git a/include/kvm/arm_pmu.h b/include/kvm/arm_pmu.h
index a24788243ac99..2604a6a46d5f3 100644
--- a/include/kvm/arm_pmu.h
+++ b/include/kvm/arm_pmu.h
@@ -102,6 +102,9 @@ void kvm_pmu_direct_pmcr_write(struct kvm_vcpu *vcpu, u64 val);
u64 kvm_pmu_direct_pmcr_read(struct kvm_vcpu *vcpu);
u64 kvm_pmu_host_counter_mask(void);
u64 kvm_pmu_guest_counter_mask(void);
+void kvm_pmu_load(struct kvm_vcpu *vcpu);
+void kvm_pmu_put(struct kvm_vcpu *vcpu);
+
/*
* Updates the vcpu's view of the pmu events for this cpu.
* Must be called before every vcpu run after disabling interrupts, to ensure
@@ -150,6 +153,8 @@ static inline u64 kvm_pmu_direct_pmcr_read(struct kvm_vcpu *vcpu)
{
return 0;
}
+static inline void kvm_pmu_load(struct kvm_vcpu *vcpu) {}
+static inline void kvm_pmu_put(struct kvm_vcpu *vcpu) {}
static inline void kvm_pmu_set_counter_value(struct kvm_vcpu *vcpu,
u64 select_idx, u64 val) {}
static inline void kvm_pmu_set_counter_value_user(struct kvm_vcpu *vcpu,
--
2.56.0.rc1.310.g51773c2048-goog