aboutsummaryrefslogtreecommitdiff
path: root/arch
diff options
context:
space:
mode:
authorPaolo Bonzini <pbonzini@redhat.com>2026-10-01 13:13:57 -0400
committerPaolo Bonzini <pbonzini@redhat.com>2026-10-01 13:13:57 -0400
commitf9bfc323e76120c2cd1fdea93bbf315eec5eb934 (patch)
tree2dd10da50f985e08d1dc71386f9164ad2309b3c9 /arch
parent973ea70393e885e540f714904e51bc6cac80e3d7 (diff)
parent71cc2c67fb8f8d5aa8154eb482e9846d2214f11b (diff)
Merge tag 'kvmarm-fixes-7.3-2' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm into HEAD
KVM/arm64 fixes for 7.3, round #2 - Take a reference on the last IRQ loaded into an LR to prevent it from being freed while running the guest (Marc Zyngier) - Ensure that the ITS MOVALL command only affects LPIs that were previously affined to the source redistributor (Marc Zyngier) - Fix + test for honoring the host's trap configuration when running non-protected VMs while KVM is in protected mode (Fuad Tabba) - Use the host stage-1 mapping granularity for VM_PFNMAP mappings at stage-2 (Mostafa Saleh)
Diffstat (limited to 'arch')
-rw-r--r--arch/arm64/kvm/emulate-nested.c3
-rw-r--r--arch/arm64/kvm/hyp/include/nvhe/pkvm.h9
-rw-r--r--arch/arm64/kvm/hyp/nvhe/hyp-main.c7
-rw-r--r--arch/arm64/kvm/hyp/nvhe/pkvm.c25
-rw-r--r--arch/arm64/kvm/mmu.c47
-rw-r--r--arch/arm64/kvm/vgic/vgic-its.c12
-rw-r--r--arch/arm64/kvm/vgic/vgic-v2.c6
-rw-r--r--arch/arm64/kvm/vgic/vgic-v3.c6
-rw-r--r--arch/arm64/kvm/vgic/vgic.c21
9 files changed, 67 insertions, 69 deletions
diff --git a/arch/arm64/kvm/emulate-nested.c b/arch/arm64/kvm/emulate-nested.c
index 3806ff0920fe..f59d89d04eca 100644
--- a/arch/arm64/kvm/emulate-nested.c
+++ b/arch/arm64/kvm/emulate-nested.c
@@ -2386,9 +2386,6 @@ int __init populate_nv_trap_config(void)
print_nv_trap_error(fgt, "FGT bit is reserved", ret);
}
- if (!cpus_have_final_cap(ARM64_HAS_FGT))
- continue;
-
prev = xa_store(&sr_forward_xa, enc,
xa_mk_value(tc.val), GFP_KERNEL);
diff --git a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
index c904647d2f76..5c050f21066a 100644
--- a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
+++ b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
@@ -13,6 +13,15 @@
#include <nvhe/spinlock.h>
/*
+ * HCR_EL2 bits EL2 takes from the host on each entry, per VM type. The rest
+ * are EL2's own and nothing the host sets there reaches the guest.
+ */
+#define PKVM_HCR_EL2_HOST_PVM (HCR_EL2_TWI | HCR_EL2_TWE | HCR_EL2_VSE)
+#define PKVM_HCR_EL2_HOST_NPVM (PKVM_HCR_EL2_HOST_PVM | HCR_EL2_VI | HCR_EL2_VF | \
+ HCR_EL2_TVM | HCR_EL2_TID2 | HCR_EL2_TID4 | \
+ HCR_EL2_TID5 | HCR_EL2_TTLBOS)
+
+/*
* Holds the relevant data for maintaining the vcpu state completely at hyp.
*/
struct pkvm_hyp_vcpu {
diff --git a/arch/arm64/kvm/hyp/nvhe/hyp-main.c b/arch/arm64/kvm/hyp/nvhe/hyp-main.c
index 9a3b92e626ad..ac64a036b0a9 100644
--- a/arch/arm64/kvm/hyp/nvhe/hyp-main.c
+++ b/arch/arm64/kvm/hyp/nvhe/hyp-main.c
@@ -216,6 +216,7 @@ static void sync_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
+ u64 host_hcr_mask = PKVM_HCR_EL2_HOST_PVM;
fpsimd_sve_flush();
flush_debug_state(hyp_vcpu);
@@ -228,6 +229,7 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
if (vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY))
flush_hyp_vcpu_state(hyp_vcpu);
+ host_hcr_mask = PKVM_HCR_EL2_HOST_NPVM;
} else {
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
}
@@ -241,9 +243,8 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
* trap-control bit, so it must flow to the hyp vCPU alongside TWI/TWE
* for the vSError to be delivered. sync_hyp_vcpu() reflects it back.
*/
- hyp_vcpu->vcpu.arch.hcr_el2 &= ~(HCR_TWI | HCR_TWE | HCR_VSE);
- hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) &
- (HCR_TWI | HCR_TWE | HCR_VSE);
+ hyp_vcpu->vcpu.arch.hcr_el2 &= ~host_hcr_mask;
+ hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) & host_hcr_mask;
hyp_vcpu->vcpu.arch.iflags = host_vcpu->arch.iflags;
diff --git a/arch/arm64/kvm/hyp/nvhe/pkvm.c b/arch/arm64/kvm/hyp/nvhe/pkvm.c
index bb3e0dc0676e..6290c4b62659 100644
--- a/arch/arm64/kvm/hyp/nvhe/pkvm.c
+++ b/arch/arm64/kvm/hyp/nvhe/pkvm.c
@@ -30,6 +30,7 @@ unsigned int kvm_host_sve_max_vl;
*/
static DEFINE_PER_CPU(struct pkvm_hyp_vcpu *, loaded_hyp_vcpu);
+/* The PKVM_HCR_EL2_HOST_{PVM,NPVM} bits of this value come from the host on each entry. */
static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu)
{
vcpu->arch.hcr_el2 = HCR_GUEST_FLAGS;
@@ -47,18 +48,17 @@ static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu)
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
vcpu->arch.hcr_el2 |= HCR_FWB;
- if (cpus_have_final_cap(ARM64_HAS_EVT) &&
- !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) &&
- kvm_read_vm_id_reg(vcpu->kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0))
- vcpu->arch.hcr_el2 |= HCR_TID4;
- else
- vcpu->arch.hcr_el2 |= HCR_TID2;
+ /*
+ * Without AArch32 EL1, leave RW set and let the entry fail with an
+ * illegal exception return: the *32_EL2 registers EL2 would otherwise
+ * switch are UNDEFINED there.
+ */
+ if (vcpu_has_feature(vcpu, KVM_ARM_VCPU_EL1_32BIT) &&
+ cpus_have_final_cap(ARM64_HAS_32BIT_EL1))
+ vcpu->arch.hcr_el2 &= ~HCR_EL2_RW;
if (vcpu_has_ptrauth(vcpu))
vcpu->arch.hcr_el2 |= (HCR_API | HCR_APK);
-
- if (kvm_has_mte(vcpu->kvm))
- vcpu->arch.hcr_el2 |= HCR_ATA;
}
static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu)
@@ -76,6 +76,13 @@ static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu)
*/
val |= HCR_TACR | HCR_TIDCP | HCR_TID3 | HCR_TID1;
+ if (cpus_have_final_cap(ARM64_HAS_EVT) &&
+ !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) &&
+ kvm_read_vm_id_reg(kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0))
+ val |= HCR_EL2_TID4;
+ else
+ val |= HCR_EL2_TID2;
+
if (!kvm_has_feat(kvm, ID_AA64PFR0_EL1, RAS, IMP)) {
val |= HCR_TERR | HCR_TEA;
val &= ~(HCR_FIEN);
diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
index 2d44cd6a5aed..bf6d6526639f 100644
--- a/arch/arm64/kvm/mmu.c
+++ b/arch/arm64/kvm/mmu.c
@@ -1468,32 +1468,11 @@ transparent_hugepage_adjust(struct kvm *kvm, struct kvm_memory_slot *memslot,
return PAGE_SIZE;
}
-static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva)
+static int get_vma_page_shift(struct vm_area_struct *vma)
{
- unsigned long pa;
-
- if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP))
+ if (is_vm_hugetlb_page(vma))
return huge_page_shift(hstate_vma(vma));
- if (!(vma->vm_flags & VM_PFNMAP))
- return PAGE_SHIFT;
-
- VM_BUG_ON(is_vm_hugetlb_page(vma));
-
- pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start);
-
-#ifndef __PAGETABLE_PMD_FOLDED
- if ((hva & (PUD_SIZE - 1)) == (pa & (PUD_SIZE - 1)) &&
- ALIGN_DOWN(hva, PUD_SIZE) >= vma->vm_start &&
- ALIGN(hva, PUD_SIZE) <= vma->vm_end)
- return PUD_SHIFT;
-#endif
-
- if ((hva & (PMD_SIZE - 1)) == (pa & (PMD_SIZE - 1)) &&
- ALIGN_DOWN(hva, PMD_SIZE) >= vma->vm_start &&
- ALIGN(hva, PMD_SIZE) <= vma->vm_end)
- return PMD_SHIFT;
-
return PAGE_SHIFT;
}
@@ -1794,7 +1773,7 @@ static short kvm_s2_resolve_vma_size(const struct kvm_s2_fault_desc *s2fd,
vma_shift = PAGE_SHIFT;
} else {
s2vi->max_map_size = PUD_SIZE;
- vma_shift = get_vma_page_shift(vma, s2fd->hva);
+ vma_shift = get_vma_page_shift(vma);
}
switch (vma_shift) {
@@ -1952,16 +1931,6 @@ static int kvm_s2_fault_pin_pfn(const struct kvm_s2_fault_desc *s2fd,
return -EFAULT;
}
} else {
- /*
- * If the page was identified as device early by looking at
- * the VMA flags, vma_pagesize is already representing the
- * largest quantity we can map. If instead it was mapped
- * via __kvm_faultin_pfn(), vma_pagesize is set to PAGE_SIZE
- * and must not be upgraded.
- *
- * In both cases, we don't let transparent_hugepage_adjust()
- * change things at the last minute.
- */
s2vi->map_non_cacheable = true;
}
@@ -2051,10 +2020,10 @@ static int kvm_s2_fault_map(const struct kvm_s2_fault_desc *s2fd,
/*
* If we are not forced to use page mapping, check if we are
- * backed by a THP and thus use block mapping if possible.
+ * backed by a huge stage-1 mapping and thus use block mapping if
+ * possible.
*/
- if (mapping_size == PAGE_SIZE &&
- !(s2vi->max_map_size == PAGE_SIZE || s2vi->map_non_cacheable)) {
+ if (mapping_size == PAGE_SIZE && s2vi->max_map_size != PAGE_SIZE) {
if (perm_fault_granule > PAGE_SIZE) {
mapping_size = perm_fault_granule;
} else {
@@ -2135,10 +2104,6 @@ static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
return ret;
}
- /*
- * Let's check if we will get back a huge page backed by hugetlbfs, or
- * get block mapping for device MMIO region.
- */
ret = kvm_s2_fault_pin_pfn(s2fd, &s2vi);
if (ret != 1)
return ret;
diff --git a/arch/arm64/kvm/vgic/vgic-its.c b/arch/arm64/kvm/vgic/vgic-its.c
index 0904ae850c35..c10a11373e10 100644
--- a/arch/arm64/kvm/vgic/vgic-its.c
+++ b/arch/arm64/kvm/vgic/vgic-its.c
@@ -319,12 +319,16 @@ static int update_lpi_config(struct kvm *kvm, struct vgic_irq *irq,
return ret;
}
-static int update_affinity(struct vgic_irq *irq, struct kvm_vcpu *vcpu)
+static int update_affinity(struct vgic_irq *irq,
+ struct kvm_vcpu *from_vcpu, struct kvm_vcpu *vcpu)
{
struct its_vlpi_map map;
int ret;
guard(raw_spinlock_irqsave)(&irq->irq_lock);
+ if (from_vcpu && irq->target_vcpu != from_vcpu)
+ return 0;
+
irq->target_vcpu = vcpu;
if (!irq->hw)
@@ -362,7 +366,7 @@ static void update_affinity_ite(struct kvm *kvm, struct its_ite *ite)
return;
vcpu = collection_to_vcpu(kvm, ite->collection);
- update_affinity(ite->irq, vcpu);
+ update_affinity(ite->irq, NULL, vcpu);
}
/*
@@ -856,7 +860,7 @@ static int vgic_its_cmd_handle_movi(struct kvm *kvm, struct vgic_its *its,
vgic_its_invalidate_cache(its);
- return update_affinity(ite->irq, vcpu);
+ return update_affinity(ite->irq, NULL, vcpu);
}
static bool __is_visible_gfn_locked(struct vgic_its *its, gpa_t gpa)
@@ -1383,7 +1387,7 @@ static int vgic_its_cmd_handle_movall(struct kvm *kvm, struct vgic_its *its,
if (!irq)
continue;
- update_affinity(irq, vcpu2);
+ update_affinity(irq, vcpu1, vcpu2);
vgic_put_irq(kvm, irq);
}
diff --git a/arch/arm64/kvm/vgic/vgic-v2.c b/arch/arm64/kvm/vgic/vgic-v2.c
index 7182f63fc938..70cc53ba810a 100644
--- a/arch/arm64/kvm/vgic/vgic-v2.c
+++ b/arch/arm64/kvm/vgic/vgic-v2.c
@@ -122,6 +122,10 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu)
for (int lr = 0; lr < vgic_cpu->vgic_v2.used_lrs; lr++)
vgic_v2_fold_lr(vcpu, cpuif->vgic_lr[lr]);
+ cpuif->used_lrs = 0;
+ if (!irq)
+ return;
+
/* See the GICv3 equivalent for the EOIcount handling rationale */
list_for_each_entry_continue(irq, &vgic_cpu->ap_list_head, ap_list) {
u32 lr;
@@ -144,8 +148,6 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu)
vgic_v2_fold_lr(vcpu, lr);
eoicount--;
}
-
- cpuif->used_lrs = 0;
}
void vgic_v2_deactivate(struct kvm_vcpu *vcpu, u32 val)
diff --git a/arch/arm64/kvm/vgic/vgic-v3.c b/arch/arm64/kvm/vgic/vgic-v3.c
index 726e20a1da6e..346bacb3198f 100644
--- a/arch/arm64/kvm/vgic/vgic-v3.c
+++ b/arch/arm64/kvm/vgic/vgic-v3.c
@@ -155,6 +155,10 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu)
for (int lr = 0; lr < cpuif->used_lrs; lr++)
vgic_v3_fold_lr(vcpu, cpuif->vgic_lr[lr]);
+ cpuif->used_lrs = 0;
+ if (!irq)
+ return;
+
/*
* EOIMode=0: use EOIcount to emulate deactivation. We are
* guaranteed to deactivate in reverse order of the activation, so
@@ -188,8 +192,6 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu)
vgic_v3_fold_lr(vcpu, lr);
eoicount--;
}
-
- cpuif->used_lrs = 0;
}
void vgic_v3_deactivate(struct kvm_vcpu *vcpu, u64 val)
diff --git a/arch/arm64/kvm/vgic/vgic.c b/arch/arm64/kvm/vgic/vgic.c
index b25303d9919f..5cf5a1ef86cd 100644
--- a/arch/arm64/kvm/vgic/vgic.c
+++ b/arch/arm64/kvm/vgic/vgic.c
@@ -853,6 +853,17 @@ retry:
goto retry;
}
+ /*
+ * Fix the last_lr_irq refcount which was obtained while
+ * populating the LRs. This can also result in the LPI being
+ * deleted.
+ */
+ irq = *host_data_ptr(last_lr_irq);
+ if (irq) {
+ deleted_lpis |= vgic_put_irq_norelease(vcpu->kvm, irq);
+ *host_data_ptr(last_lr_irq) = NULL;
+ }
+
raw_spin_unlock(&vgic_cpu->ap_list_lock);
if (unlikely(deleted_lpis))
@@ -866,9 +877,6 @@ static void vgic_fold_state(struct kvm_vcpu *vcpu)
return;
}
- if (!*host_data_ptr(last_lr_irq))
- return;
-
if (kvm_vgic_global_state.type == VGIC_V2)
vgic_v2_fold_lr_state(vcpu);
else
@@ -1021,11 +1029,14 @@ static void vgic_flush_lr_state(struct kvm_vcpu *vcpu)
scoped_guard(raw_spinlock, &irq->irq_lock) {
if (likely(vgic_target_oracle(irq) == vcpu)) {
vgic_populate_lr(vcpu, irq, count++);
- *host_data_ptr(last_lr_irq) = irq;
+ if (count == kvm_vgic_global_state.nr_lr) {
+ vgic_get_irq_ref(irq);
+ *host_data_ptr(last_lr_irq) = irq;
+ }
}
}
- if (count == kvm_vgic_global_state.nr_lr)
+ if (*host_data_ptr(last_lr_irq))
break;
}