summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-01 10:57:19 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-01 10:57:19 -0700
commita940b03cee1524c10c16e0f73ec878bbd36202a2 (patch)
tree931c568ee3aa3bbab3da762e80adc54c4f4bce0a
parentd24e8ac715de2e16a53c144005b1863660a5fbea (diff)
parentf9bfc323e76120c2cd1fdea93bbf315eec5eb934 (diff)
Merge tag 'for-linus' of git://git.kernel.org/pub/scm/virt/kvm/kvm
Pull kvm fixes from Paolo Bonzini: "The most intrusive change is reverting a commit from 7.3-rc1 that made struct kvm a bit too large, and fixing the same issue otherwise. There are again a lot of selftests lines; the sheer number of commits is not small but I don't expect much more for 7.3 due to people travelling to Plumbers next week. ARM: - Take a reference on the last IRQ loaded into an LR to prevent it from being freed while running the guest (Marc Zyngier) - Ensure that the ITS MOVALL command only affects LPIs that were previously affined to the source redistributor (Marc Zyngier) - Fix + test for honoring the host's trap configuration when running non-protected VMs while KVM is in protected mode (Fuad Tabba) - Use the host stage-1 mapping granularity for VM_PFNMAP mappings at stage-2 (Mostafa Saleh) x86: Various bugfixes where the guest could do stupid things on purpose to cause problems in the host: - Failed VMRUNs can cause pending TLB flushes to be dropped, and in general some actions done through VMCB control fields have to be redone if VMRUN fails - Toggling MSR interceptions or eVMCS execution controls can cause the host to use a stale MSR permission bitmap - Bad page tables can cause a WARN. Also fix issues in last week's pull request (my fault, for changing email workflow and thus missing feedback sent to kvm@ but not LKML). Generic: - Take kvm_lock when creating vCPUs. For almost two decades everybody thought it was not done for some unspecified performance reasons, but in reality it was only done because kvm_lock was originally a spinlock. This is a better fix than 97d65b544f48 ("KVM: Check for duplicate vcpu_id as early as possible", from the 7.3 merge window), and does not waste 2K per VM, hence its inclusion here" * tag 'for-linus' of git://git.kernel.org/pub/scm/virt/kvm/kvm: (29 commits) KVM: arm64: Use stage-1 leaf size for VM_PFNMAP KVM: arm64: selftests: Check a feature hidden in an ID register is UNDEF KVM: arm64: Use the host's HCR_EL2 for non-protected VMs in pKVM KVM: arm64: Clear HCR_EL2.RW for 32-bit non-protected vCPUs KVM: arm64: Apply the fine-grained UNDEFs without FEAT_FGT KVM: arm64: vgic-its: Fix MOVALL handling of source redistributor KVM: arm64: vgic: Take a refcount on IRQs referenced by last_lr_irq KVM: arm64: vgic: Allow last_lr_irq to be NULL when LRs are not overflowing KVM: SEV: Do cache maintenance on the source VM *before* clearing SEV state KVM: SEV: Nullify "have run CPUs" mask pointer when freeing it KVM: selftests: Extend nested x2APIC test to validate using eVMCS for vmcs12 KVM: selftests: Extend nested x2APIC test to validate disabling x2APIC virt KVM: selftests: Verify that L0's TPR doesn't get clobbered KVM: selftests: Run the nested x2APIC with and without APICv being inhibited in L2 KVM: selftests: Add x2APIC MSR test for inhibiting APICv while nested KVM: nVMX: Force MSR bitmap refresh if runtime eVMCS controls are modified KVM: SVM: Use the active VMCB's MSR bitmap when checking if MSR is intercepted KVM: SVM: Sync guest's PERF_CNTR_GLOBAL_CTL from h/w only on successful VMRUN KVM: SVM: Don't mark ASID fields as dirty when setting control.tlb_ctl KVM: SVM: Update control fields on #VMEXIT if and only if VMRUN succeeded ...
-rw-r--r--arch/arm64/kvm/emulate-nested.c3
-rw-r--r--arch/arm64/kvm/hyp/include/nvhe/pkvm.h9
-rw-r--r--arch/arm64/kvm/hyp/nvhe/hyp-main.c7
-rw-r--r--arch/arm64/kvm/hyp/nvhe/pkvm.c25
-rw-r--r--arch/arm64/kvm/mmu.c47
-rw-r--r--arch/arm64/kvm/vgic/vgic-init.c12
-rw-r--r--arch/arm64/kvm/vgic/vgic-its.c12
-rw-r--r--arch/arm64/kvm/vgic/vgic-v2.c6
-rw-r--r--arch/arm64/kvm/vgic/vgic-v3.c6
-rw-r--r--arch/arm64/kvm/vgic/vgic.c21
-rw-r--r--arch/powerpc/kvm/book3s_hv.c2
-rw-r--r--arch/riscv/kvm/aia_device.c2
-rw-r--r--arch/s390/kvm/s390/s390.c5
-rw-r--r--arch/x86/kvm/mmu/mmu.c6
-rw-r--r--arch/x86/kvm/svm/sev.c57
-rw-r--r--arch/x86/kvm/svm/svm.c43
-rw-r--r--arch/x86/kvm/svm/svm.h1
-rw-r--r--arch/x86/kvm/vmx/nested.c6
-rw-r--r--arch/x86/kvm/vmx/tdx.c5
-rw-r--r--include/linux/kvm_host.h1
-rw-r--r--tools/testing/selftests/kvm/Makefile.kvm2
-rw-r--r--tools/testing/selftests/kvm/arm64/hidden_features.c184
-rw-r--r--tools/testing/selftests/kvm/x86/nested_x2apic_test.c222
-rw-r--r--virt/kvm/kvm_main.c35
24 files changed, 554 insertions, 165 deletions
diff --git a/arch/arm64/kvm/emulate-nested.c b/arch/arm64/kvm/emulate-nested.c
index 3806ff0920fe..f59d89d04eca 100644
--- a/arch/arm64/kvm/emulate-nested.c
+++ b/arch/arm64/kvm/emulate-nested.c
@@ -2386,9 +2386,6 @@ int __init populate_nv_trap_config(void)
print_nv_trap_error(fgt, "FGT bit is reserved", ret);
}
- if (!cpus_have_final_cap(ARM64_HAS_FGT))
- continue;
-
prev = xa_store(&sr_forward_xa, enc,
xa_mk_value(tc.val), GFP_KERNEL);
diff --git a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
index c904647d2f76..5c050f21066a 100644
--- a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
+++ b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h
@@ -13,6 +13,15 @@
#include <nvhe/spinlock.h>
/*
+ * HCR_EL2 bits EL2 takes from the host on each entry, per VM type. The rest
+ * are EL2's own and nothing the host sets there reaches the guest.
+ */
+#define PKVM_HCR_EL2_HOST_PVM (HCR_EL2_TWI | HCR_EL2_TWE | HCR_EL2_VSE)
+#define PKVM_HCR_EL2_HOST_NPVM (PKVM_HCR_EL2_HOST_PVM | HCR_EL2_VI | HCR_EL2_VF | \
+ HCR_EL2_TVM | HCR_EL2_TID2 | HCR_EL2_TID4 | \
+ HCR_EL2_TID5 | HCR_EL2_TTLBOS)
+
+/*
* Holds the relevant data for maintaining the vcpu state completely at hyp.
*/
struct pkvm_hyp_vcpu {
diff --git a/arch/arm64/kvm/hyp/nvhe/hyp-main.c b/arch/arm64/kvm/hyp/nvhe/hyp-main.c
index 9a3b92e626ad..ac64a036b0a9 100644
--- a/arch/arm64/kvm/hyp/nvhe/hyp-main.c
+++ b/arch/arm64/kvm/hyp/nvhe/hyp-main.c
@@ -216,6 +216,7 @@ static void sync_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu)
static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
{
struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu;
+ u64 host_hcr_mask = PKVM_HCR_EL2_HOST_PVM;
fpsimd_sve_flush();
flush_debug_state(hyp_vcpu);
@@ -228,6 +229,7 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) {
if (vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY))
flush_hyp_vcpu_state(hyp_vcpu);
+ host_hcr_mask = PKVM_HCR_EL2_HOST_NPVM;
} else {
hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt;
}
@@ -241,9 +243,8 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu)
* trap-control bit, so it must flow to the hyp vCPU alongside TWI/TWE
* for the vSError to be delivered. sync_hyp_vcpu() reflects it back.
*/
- hyp_vcpu->vcpu.arch.hcr_el2 &= ~(HCR_TWI | HCR_TWE | HCR_VSE);
- hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) &
- (HCR_TWI | HCR_TWE | HCR_VSE);
+ hyp_vcpu->vcpu.arch.hcr_el2 &= ~host_hcr_mask;
+ hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) & host_hcr_mask;
hyp_vcpu->vcpu.arch.iflags = host_vcpu->arch.iflags;
diff --git a/arch/arm64/kvm/hyp/nvhe/pkvm.c b/arch/arm64/kvm/hyp/nvhe/pkvm.c
index bb3e0dc0676e..6290c4b62659 100644
--- a/arch/arm64/kvm/hyp/nvhe/pkvm.c
+++ b/arch/arm64/kvm/hyp/nvhe/pkvm.c
@@ -30,6 +30,7 @@ unsigned int kvm_host_sve_max_vl;
*/
static DEFINE_PER_CPU(struct pkvm_hyp_vcpu *, loaded_hyp_vcpu);
+/* The PKVM_HCR_EL2_HOST_{PVM,NPVM} bits of this value come from the host on each entry. */
static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu)
{
vcpu->arch.hcr_el2 = HCR_GUEST_FLAGS;
@@ -47,18 +48,17 @@ static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu)
if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB))
vcpu->arch.hcr_el2 |= HCR_FWB;
- if (cpus_have_final_cap(ARM64_HAS_EVT) &&
- !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) &&
- kvm_read_vm_id_reg(vcpu->kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0))
- vcpu->arch.hcr_el2 |= HCR_TID4;
- else
- vcpu->arch.hcr_el2 |= HCR_TID2;
+ /*
+ * Without AArch32 EL1, leave RW set and let the entry fail with an
+ * illegal exception return: the *32_EL2 registers EL2 would otherwise
+ * switch are UNDEFINED there.
+ */
+ if (vcpu_has_feature(vcpu, KVM_ARM_VCPU_EL1_32BIT) &&
+ cpus_have_final_cap(ARM64_HAS_32BIT_EL1))
+ vcpu->arch.hcr_el2 &= ~HCR_EL2_RW;
if (vcpu_has_ptrauth(vcpu))
vcpu->arch.hcr_el2 |= (HCR_API | HCR_APK);
-
- if (kvm_has_mte(vcpu->kvm))
- vcpu->arch.hcr_el2 |= HCR_ATA;
}
static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu)
@@ -76,6 +76,13 @@ static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu)
*/
val |= HCR_TACR | HCR_TIDCP | HCR_TID3 | HCR_TID1;
+ if (cpus_have_final_cap(ARM64_HAS_EVT) &&
+ !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) &&
+ kvm_read_vm_id_reg(kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0))
+ val |= HCR_EL2_TID4;
+ else
+ val |= HCR_EL2_TID2;
+
if (!kvm_has_feat(kvm, ID_AA64PFR0_EL1, RAS, IMP)) {
val |= HCR_TERR | HCR_TEA;
val &= ~(HCR_FIEN);
diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
index 2d44cd6a5aed..bf6d6526639f 100644
--- a/arch/arm64/kvm/mmu.c
+++ b/arch/arm64/kvm/mmu.c
@@ -1468,32 +1468,11 @@ transparent_hugepage_adjust(struct kvm *kvm, struct kvm_memory_slot *memslot,
return PAGE_SIZE;
}
-static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva)
+static int get_vma_page_shift(struct vm_area_struct *vma)
{
- unsigned long pa;
-
- if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP))
+ if (is_vm_hugetlb_page(vma))
return huge_page_shift(hstate_vma(vma));
- if (!(vma->vm_flags & VM_PFNMAP))
- return PAGE_SHIFT;
-
- VM_BUG_ON(is_vm_hugetlb_page(vma));
-
- pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start);
-
-#ifndef __PAGETABLE_PMD_FOLDED
- if ((hva & (PUD_SIZE - 1)) == (pa & (PUD_SIZE - 1)) &&
- ALIGN_DOWN(hva, PUD_SIZE) >= vma->vm_start &&
- ALIGN(hva, PUD_SIZE) <= vma->vm_end)
- return PUD_SHIFT;
-#endif
-
- if ((hva & (PMD_SIZE - 1)) == (pa & (PMD_SIZE - 1)) &&
- ALIGN_DOWN(hva, PMD_SIZE) >= vma->vm_start &&
- ALIGN(hva, PMD_SIZE) <= vma->vm_end)
- return PMD_SHIFT;
-
return PAGE_SHIFT;
}
@@ -1794,7 +1773,7 @@ static short kvm_s2_resolve_vma_size(const struct kvm_s2_fault_desc *s2fd,
vma_shift = PAGE_SHIFT;
} else {
s2vi->max_map_size = PUD_SIZE;
- vma_shift = get_vma_page_shift(vma, s2fd->hva);
+ vma_shift = get_vma_page_shift(vma);
}
switch (vma_shift) {
@@ -1952,16 +1931,6 @@ static int kvm_s2_fault_pin_pfn(const struct kvm_s2_fault_desc *s2fd,
return -EFAULT;
}
} else {
- /*
- * If the page was identified as device early by looking at
- * the VMA flags, vma_pagesize is already representing the
- * largest quantity we can map. If instead it was mapped
- * via __kvm_faultin_pfn(), vma_pagesize is set to PAGE_SIZE
- * and must not be upgraded.
- *
- * In both cases, we don't let transparent_hugepage_adjust()
- * change things at the last minute.
- */
s2vi->map_non_cacheable = true;
}
@@ -2051,10 +2020,10 @@ static int kvm_s2_fault_map(const struct kvm_s2_fault_desc *s2fd,
/*
* If we are not forced to use page mapping, check if we are
- * backed by a THP and thus use block mapping if possible.
+ * backed by a huge stage-1 mapping and thus use block mapping if
+ * possible.
*/
- if (mapping_size == PAGE_SIZE &&
- !(s2vi->max_map_size == PAGE_SIZE || s2vi->map_non_cacheable)) {
+ if (mapping_size == PAGE_SIZE && s2vi->max_map_size != PAGE_SIZE) {
if (perm_fault_granule > PAGE_SIZE) {
mapping_size = perm_fault_granule;
} else {
@@ -2135,10 +2104,6 @@ static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
return ret;
}
- /*
- * Let's check if we will get back a huge page backed by hugetlbfs, or
- * get block mapping for device MMIO region.
- */
ret = kvm_s2_fault_pin_pfn(s2fd, &s2vi);
if (ret != 1)
return ret;
diff --git a/arch/arm64/kvm/vgic/vgic-init.c b/arch/arm64/kvm/vgic/vgic-init.c
index 4012df6002ea..a58575df36e9 100644
--- a/arch/arm64/kvm/vgic/vgic-init.c
+++ b/arch/arm64/kvm/vgic/vgic-init.c
@@ -97,6 +97,9 @@ int kvm_vgic_create(struct kvm *kvm, u32 type)
/*
* - Acquiring the vCPU mutex for every *online* vCPU to prevent
* concurrent vCPU ioctls for vCPUs already visible to userspace.
+ * This also ensures KVM isn't in the middle of creating a vCPU,
+ * i.e. that there are no vCPUs that have been created but aren't
+ * yet fully online.
*/
ret = -EBUSY;
if (kvm_trylock_all_vcpus(kvm))
@@ -105,18 +108,11 @@ int kvm_vgic_create(struct kvm *kvm, u32 type)
/*
* - Taking the config_lock which protects VGIC data structures such
* as the per-vCPU arrays of private IRQs (SGIs, PPIs).
- */
- mutex_lock(&kvm->arch.config_lock);
-
- /*
- * - Bailing on the entire thing if a vCPU is in the middle of creation,
- * dropped the kvm->lock, but hasn't reached kvm_arch_vcpu_create().
*
* The whole combination of this guarantees that no vCPU can get into
* KVM with a VGIC configuration inconsistent with the VM's VGIC.
*/
- if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus))
- goto out_unlock;
+ mutex_lock(&kvm->arch.config_lock);
if (irqchip_in_kernel(kvm)) {
ret = -EEXIST;
diff --git a/arch/arm64/kvm/vgic/vgic-its.c b/arch/arm64/kvm/vgic/vgic-its.c
index 0904ae850c35..c10a11373e10 100644
--- a/arch/arm64/kvm/vgic/vgic-its.c
+++ b/arch/arm64/kvm/vgic/vgic-its.c
@@ -319,12 +319,16 @@ static int update_lpi_config(struct kvm *kvm, struct vgic_irq *irq,
return ret;
}
-static int update_affinity(struct vgic_irq *irq, struct kvm_vcpu *vcpu)
+static int update_affinity(struct vgic_irq *irq,
+ struct kvm_vcpu *from_vcpu, struct kvm_vcpu *vcpu)
{
struct its_vlpi_map map;
int ret;
guard(raw_spinlock_irqsave)(&irq->irq_lock);
+ if (from_vcpu && irq->target_vcpu != from_vcpu)
+ return 0;
+
irq->target_vcpu = vcpu;
if (!irq->hw)
@@ -362,7 +366,7 @@ static void update_affinity_ite(struct kvm *kvm, struct its_ite *ite)
return;
vcpu = collection_to_vcpu(kvm, ite->collection);
- update_affinity(ite->irq, vcpu);
+ update_affinity(ite->irq, NULL, vcpu);
}
/*
@@ -856,7 +860,7 @@ static int vgic_its_cmd_handle_movi(struct kvm *kvm, struct vgic_its *its,
vgic_its_invalidate_cache(its);
- return update_affinity(ite->irq, vcpu);
+ return update_affinity(ite->irq, NULL, vcpu);
}
static bool __is_visible_gfn_locked(struct vgic_its *its, gpa_t gpa)
@@ -1383,7 +1387,7 @@ static int vgic_its_cmd_handle_movall(struct kvm *kvm, struct vgic_its *its,
if (!irq)
continue;
- update_affinity(irq, vcpu2);
+ update_affinity(irq, vcpu1, vcpu2);
vgic_put_irq(kvm, irq);
}
diff --git a/arch/arm64/kvm/vgic/vgic-v2.c b/arch/arm64/kvm/vgic/vgic-v2.c
index 7182f63fc938..70cc53ba810a 100644
--- a/arch/arm64/kvm/vgic/vgic-v2.c
+++ b/arch/arm64/kvm/vgic/vgic-v2.c
@@ -122,6 +122,10 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu)
for (int lr = 0; lr < vgic_cpu->vgic_v2.used_lrs; lr++)
vgic_v2_fold_lr(vcpu, cpuif->vgic_lr[lr]);
+ cpuif->used_lrs = 0;
+ if (!irq)
+ return;
+
/* See the GICv3 equivalent for the EOIcount handling rationale */
list_for_each_entry_continue(irq, &vgic_cpu->ap_list_head, ap_list) {
u32 lr;
@@ -144,8 +148,6 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu)
vgic_v2_fold_lr(vcpu, lr);
eoicount--;
}
-
- cpuif->used_lrs = 0;
}
void vgic_v2_deactivate(struct kvm_vcpu *vcpu, u32 val)
diff --git a/arch/arm64/kvm/vgic/vgic-v3.c b/arch/arm64/kvm/vgic/vgic-v3.c
index 726e20a1da6e..346bacb3198f 100644
--- a/arch/arm64/kvm/vgic/vgic-v3.c
+++ b/arch/arm64/kvm/vgic/vgic-v3.c
@@ -155,6 +155,10 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu)
for (int lr = 0; lr < cpuif->used_lrs; lr++)
vgic_v3_fold_lr(vcpu, cpuif->vgic_lr[lr]);
+ cpuif->used_lrs = 0;
+ if (!irq)
+ return;
+
/*
* EOIMode=0: use EOIcount to emulate deactivation. We are
* guaranteed to deactivate in reverse order of the activation, so
@@ -188,8 +192,6 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu)
vgic_v3_fold_lr(vcpu, lr);
eoicount--;
}
-
- cpuif->used_lrs = 0;
}
void vgic_v3_deactivate(struct kvm_vcpu *vcpu, u64 val)
diff --git a/arch/arm64/kvm/vgic/vgic.c b/arch/arm64/kvm/vgic/vgic.c
index b25303d9919f..5cf5a1ef86cd 100644
--- a/arch/arm64/kvm/vgic/vgic.c
+++ b/arch/arm64/kvm/vgic/vgic.c
@@ -853,6 +853,17 @@ retry:
goto retry;
}
+ /*
+ * Fix the last_lr_irq refcount which was obtained while
+ * populating the LRs. This can also result in the LPI being
+ * deleted.
+ */
+ irq = *host_data_ptr(last_lr_irq);
+ if (irq) {
+ deleted_lpis |= vgic_put_irq_norelease(vcpu->kvm, irq);
+ *host_data_ptr(last_lr_irq) = NULL;
+ }
+
raw_spin_unlock(&vgic_cpu->ap_list_lock);
if (unlikely(deleted_lpis))
@@ -866,9 +877,6 @@ static void vgic_fold_state(struct kvm_vcpu *vcpu)
return;
}
- if (!*host_data_ptr(last_lr_irq))
- return;
-
if (kvm_vgic_global_state.type == VGIC_V2)
vgic_v2_fold_lr_state(vcpu);
else
@@ -1021,11 +1029,14 @@ static void vgic_flush_lr_state(struct kvm_vcpu *vcpu)
scoped_guard(raw_spinlock, &irq->irq_lock) {
if (likely(vgic_target_oracle(irq) == vcpu)) {
vgic_populate_lr(vcpu, irq, count++);
- *host_data_ptr(last_lr_irq) = irq;
+ if (count == kvm_vgic_global_state.nr_lr) {
+ vgic_get_irq_ref(irq);
+ *host_data_ptr(last_lr_irq) = irq;
+ }
}
}
- if (count == kvm_vgic_global_state.nr_lr)
+ if (*host_data_ptr(last_lr_irq))
break;
}
diff --git a/arch/powerpc/kvm/book3s_hv.c b/arch/powerpc/kvm/book3s_hv.c
index dbac3573b2c8..de05d721edc4 100644
--- a/arch/powerpc/kvm/book3s_hv.c
+++ b/arch/powerpc/kvm/book3s_hv.c
@@ -3058,7 +3058,6 @@ static int kvmppc_core_vcpu_create_hv(struct kvm_vcpu *vcpu)
init_waitqueue_head(&vcpu->arch.cpu_run);
- mutex_lock(&kvm->lock);
vcore = NULL;
err = -EINVAL;
if (cpu_has_feature(CPU_FTR_ARCH_300)) {
@@ -3091,7 +3090,6 @@ static int kvmppc_core_vcpu_create_hv(struct kvm_vcpu *vcpu)
mutex_unlock(&kvm->arch.mmu_setup_lock);
}
}
- mutex_unlock(&kvm->lock);
if (!vcore)
return err;
diff --git a/arch/riscv/kvm/aia_device.c b/arch/riscv/kvm/aia_device.c
index efc7c0bcfba9..97833a04268a 100644
--- a/arch/riscv/kvm/aia_device.c
+++ b/arch/riscv/kvm/aia_device.c
@@ -237,7 +237,7 @@ static int aia_init(struct kvm *kvm)
return -EBUSY;
/* We might be in the middle of creating a VCPU? */
- if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus))
+ if (kvm_is_vcpu_creation_in_progress(kvm))
return -EBUSY;
/* Number of sources should be less than or equals number of IDs */
diff --git a/arch/s390/kvm/s390/s390.c b/arch/s390/kvm/s390/s390.c
index efd77b9cd36c..e4912db67adc 100644
--- a/arch/s390/kvm/s390/s390.c
+++ b/arch/s390/kvm/s390/s390.c
@@ -3579,12 +3579,11 @@ void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu)
void kvm_arch_vcpu_postcreate(struct kvm_vcpu *vcpu)
{
- mutex_lock(&vcpu->kvm->lock);
preempt_disable();
vcpu->arch.sie_block->epoch = vcpu->kvm->arch.epoch;
vcpu->arch.sie_block->epdx = vcpu->kvm->arch.epdx;
preempt_enable();
- mutex_unlock(&vcpu->kvm->lock);
+
if (!kvm_is_ucontrol(vcpu->kvm)) {
vcpu->arch.gmap = vcpu->kvm->arch.gmap;
sca_add_vcpu(vcpu);
@@ -3757,13 +3756,11 @@ static int kvm_s390_vcpu_setup(struct kvm_vcpu *vcpu)
kvm_s390_vcpu_pci_setup(vcpu);
- mutex_lock(&vcpu->kvm->lock);
if (kvm_s390_pv_is_protected(vcpu->kvm)) {
rc = kvm_s390_pv_create_cpu(vcpu, &uvrc, &uvrrc);
if (rc)
kvm_s390_vcpu_unsetup_cmma(vcpu);
}
- mutex_unlock(&vcpu->kvm->lock);
return rc;
}
diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
index 8e62476e477b..a0d608e3fceb 100644
--- a/arch/x86/kvm/mmu/mmu.c
+++ b/arch/x86/kvm/mmu/mmu.c
@@ -2512,6 +2512,12 @@ static void shadow_walk_init_using_root(struct kvm_shadow_walk_iterator *iterato
iterator->addr = addr;
iterator->shadow_addr = root;
+
+ if (WARN_ON_ONCE(!VALID_PAGE(root)) || kvm_mmu_is_dummy_root(root)) {
+ iterator->level = 0;
+ return;
+ }
+
iterator->level = vcpu->arch.mmu->root_role.level;
if (iterator->level >= PT64_ROOT_4LEVEL &&
diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c
index 63eb2155a774..0c1ebb16cec6 100644
--- a/arch/x86/kvm/svm/sev.c
+++ b/arch/x86/kvm/svm/sev.c
@@ -488,6 +488,20 @@ static void snp_guest_req_cleanup(struct kvm *kvm)
sev->guest_resp_buf = NULL;
}
+static int sev_alloc_have_run_cpus(struct kvm_sev_info *sev)
+{
+ if (!zalloc_cpumask_var(&sev->have_run_cpus, GFP_KERNEL_ACCOUNT))
+ return -ENOMEM;
+
+ return 0;
+}
+
+static void sev_free_have_run_cpus(struct kvm_sev_info *sev)
+{
+ free_cpumask_var(sev->have_run_cpus);
+ memset(&sev->have_run_cpus, 0, sizeof(sev->have_run_cpus));
+}
+
static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp,
struct kvm_sev_init *data,
unsigned long vm_type)
@@ -545,10 +559,9 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp,
if (ret)
goto e_free_asid;
- if (!zalloc_cpumask_var(&sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) {
- ret = -ENOMEM;
+ ret = sev_alloc_have_run_cpus(sev);
+ if (ret)
goto e_free_asid;
- }
/* This needs to happen after SEV/SNP firmware initialization. */
if (snp_active) {
@@ -566,7 +579,7 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp,
return 0;
e_free:
- free_cpumask_var(sev->have_run_cpus);
+ sev_free_have_run_cpus(sev);
e_free_asid:
argp->error = init_args.error;
sev_asid_free(sev);
@@ -1125,9 +1138,6 @@ static int sev_launch_update_vmsa(struct kvm *kvm, struct kvm_sev_cmd *argp)
if (!sev_es_guest(kvm))
return -ENOTTY;
- if (kvm_is_vcpu_creation_in_progress(kvm))
- return -EBUSY;
-
ret = kvm_lock_all_vcpus(kvm);
if (ret)
return ret;
@@ -2035,6 +2045,13 @@ static void sev_migrate_from(struct kvm *dst_kvm, struct kvm *src_kvm)
struct kvm_sev_info *mirror;
unsigned long i;
+ /*
+ * Do cache maintenance on the source VM *before* clearing "SEV active",
+ * as memory reclaim flows won't trigger cache maintenance on the VM
+ * once it's no longer an SEV VM.
+ */
+ sev_writeback_caches(src_kvm);
+
dst->active = true;
dst->asid = src->asid;
dst->handle = src->handle;
@@ -2048,12 +2065,6 @@ static void sev_migrate_from(struct kvm *dst_kvm, struct kvm *src_kvm)
src->pages_locked = 0;
src->es_active = false;
- /*
- * Do cache maintenance on the source VM as it is no longer an SEV VM,
- * i.e. memory reclaim flows won't trigger cache maintenance on the VM.
- */
- sev_writeback_caches(src_kvm);
-
list_cut_before(&dst->regions_list, &src->regions_list, &src->regions_list);
mutex_lock(&sev_mirror_lock);
@@ -2121,10 +2132,6 @@ static int sev_check_source_vcpus(struct kvm *dst, struct kvm *src)
struct kvm_vcpu *src_vcpu;
unsigned long i;
- if (kvm_is_vcpu_creation_in_progress(src) ||
- kvm_is_vcpu_creation_in_progress(dst))
- return -EBUSY;
-
if (!sev_es_guest(src))
return 0;
@@ -2198,10 +2205,9 @@ int sev_vm_move_enc_context_from(struct kvm *kvm, unsigned int source_fd)
* does not, i.e. KVM could skip flushes if memory is reclaimed from
* the old VM but not the new VM.
*/
- if (!zalloc_cpumask_var(&dst_sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) {
- ret = -ENOMEM;
+ ret = sev_alloc_have_run_cpus(dst_sev);
+ if (ret)
goto out_source_vcpu;
- }
sev_migrate_from(kvm, source_kvm);
kvm_vm_dead(source_kvm);
@@ -2520,9 +2526,6 @@ static int snp_launch_update_vmsa(struct kvm *kvm, struct kvm_sev_cmd *argp)
unsigned long i;
int ret;
- if (kvm_is_vcpu_creation_in_progress(kvm))
- return -EBUSY;
-
ret = kvm_lock_all_vcpus(kvm);
if (ret)
return ret;
@@ -2898,10 +2901,9 @@ int sev_vm_copy_enc_context_from(struct kvm *kvm, unsigned int source_fd)
}
mirror_sev = to_kvm_sev_info(kvm);
- if (!zalloc_cpumask_var(&mirror_sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) {
- ret = -ENOMEM;
+ ret = sev_alloc_have_run_cpus(mirror_sev);
+ if (ret)
goto e_unlock;
- }
/*
* The mirror kvm holds an enc_context_owner ref so its asid can't
@@ -2994,7 +2996,7 @@ void sev_vm_destroy(struct kvm *kvm)
* Free the mask even if the VM is not *currently* an SEV VM, as it may
* have been an SEV VM prior to intra-host migration.
*/
- free_cpumask_var(sev->have_run_cpus);
+ sev_free_have_run_cpus(sev);
if (!sev_guest(kvm))
return;
@@ -3630,7 +3632,6 @@ int pre_sev_run(struct vcpu_svm *svm, int cpu)
sd->sev_vmcbs[asid] = svm->vmcb;
svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ASID;
- vmcb_mark_dirty(svm->vmcb, VMCB_ASID);
return 0;
}
diff --git a/arch/x86/kvm/svm/svm.c b/arch/x86/kvm/svm/svm.c
index 7d59d301e1e5..fe0cba4731b2 100644
--- a/arch/x86/kvm/svm/svm.c
+++ b/arch/x86/kvm/svm/svm.c
@@ -673,16 +673,7 @@ static void clr_dr_intercepts(struct vcpu_svm *svm)
static bool msr_write_intercepted(struct vcpu_svm *svm, u32 msr)
{
- /*
- * For non-nested case:
- * If the L01 MSR bitmap does not intercept the MSR, then we need to
- * save it.
- *
- * For nested case:
- * If the L02 MSR bitmap does not intercept the MSR, then we need to
- * save it.
- */
- void *msrpm = is_guest_mode(&svm->vcpu) ? svm->nested.msrpm : svm->msrpm;
+ void *msrpm = __va(__sme_clr(svm->vmcb->control.msrpm_base_pa));
return svm_test_msr_bitmap_write(msrpm, msr);
}
@@ -1902,8 +1893,7 @@ static void new_asid(struct vcpu_svm *svm, struct svm_cpu_data *sd)
if (sd->next_asid > sd->max_asid) {
++sd->asid_generation;
sd->next_asid = sd->min_asid;
- svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ALL_ASID;
- vmcb_mark_dirty(svm->vmcb, VMCB_ASID);
+ sd->flush_all_asids = true;
}
svm->current_vmcb->asid_generation = sd->asid_generation;
@@ -4528,6 +4518,9 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags)
svm->vmcb->control.asid = svm->asid;
vmcb_mark_dirty(svm->vmcb, VMCB_ASID);
}
+ if (this_cpu_ptr(&svm_data)->flush_all_asids)
+ svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ALL_ASID;
+
svm->vmcb->save.cr2 = vcpu->arch.cr2;
if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS) &&
@@ -4618,16 +4611,23 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags)
vcpu->arch.nested_run_pending = 0;
}
- svm->vmcb->control.tlb_ctl = TLB_CONTROL_DO_NOTHING;
+ if (!svm_is_vmrun_failure(svm->vmcb->control.exit_code)) {
+ this_cpu_ptr(&svm_data)->flush_all_asids = false;
+ svm->vmcb->control.tlb_ctl = TLB_CONTROL_DO_NOTHING;
- /*
- * Unconditionally mask off the CLEAR_RAP bit, the AND is just as cheap
- * as the TEST+Jcc to avoid it.
- */
- if (cpu_feature_enabled(X86_FEATURE_ERAPS))
- svm->vmcb->control.erap_ctl &= ~ERAP_CONTROL_CLEAR_RAP;
+ /*
+ * Unconditionally mask off the CLEAR_RAP bit, the AND is just
+ * as cheap as the TEST+Jcc to avoid it.
+ */
+ if (cpu_feature_enabled(X86_FEATURE_ERAPS))
+ svm->vmcb->control.erap_ctl &= ~ERAP_CONTROL_CLEAR_RAP;
- vmcb_mark_all_clean(svm->vmcb);
+ vmcb_mark_all_clean(svm->vmcb);
+
+ if (!msr_write_intercepted(svm, MSR_AMD64_PERF_CNTR_GLOBAL_CTL))
+ rdmsrq(MSR_AMD64_PERF_CNTR_GLOBAL_CTL,
+ vcpu_to_pmu(vcpu)->global_ctrl);
+ }
/* if exit due to PF check for async PF */
if (svm->vmcb->control.exit_code == SVM_EXIT_EXCP_BASE + PF_VECTOR)
@@ -4636,9 +4636,6 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags)
kvm_clear_available_registers(vcpu, SVM_REGS_LAZY_LOAD_SET);
- if (!msr_write_intercepted(svm, MSR_AMD64_PERF_CNTR_GLOBAL_CTL))
- rdmsrq(MSR_AMD64_PERF_CNTR_GLOBAL_CTL, vcpu_to_pmu(vcpu)->global_ctrl);
-
trace_kvm_exit(vcpu, KVM_ISA_SVM);
svm_complete_interrupts(vcpu);
diff --git a/arch/x86/kvm/svm/svm.h b/arch/x86/kvm/svm/svm.h
index e958943b8162..84f19026d3e8 100644
--- a/arch/x86/kvm/svm/svm.h
+++ b/arch/x86/kvm/svm/svm.h
@@ -376,6 +376,7 @@ struct svm_cpu_data {
u32 next_asid;
u32 min_asid;
+ bool flush_all_asids;
bool bp_spec_reduce_set;
struct vmcb *save_area;
diff --git a/arch/x86/kvm/vmx/nested.c b/arch/x86/kvm/vmx/nested.c
index 40c1a5f6fa8a..b25216862740 100644
--- a/arch/x86/kvm/vmx/nested.c
+++ b/arch/x86/kvm/vmx/nested.c
@@ -1754,6 +1754,9 @@ static void copy_vmcs12_to_shadow(struct vcpu_vmx *vmx)
static void copy_enlightened_to_vmcs12(struct vcpu_vmx *vmx, u32 hv_clean_fields)
{
#ifdef CONFIG_KVM_HYPERV
+ const u64 runtime_controls = HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_GRP1 |
+ HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_GRP2 |
+ HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_PROC;
struct vmcs12 *vmcs12 = vmx->nested.cached_vmcs12;
struct hv_enlightened_vmcs *evmcs = nested_vmx_evmcs(vmx);
struct kvm_vcpu_hv *hv_vcpu = to_hv_vcpu(&vmx->vcpu);
@@ -1762,6 +1765,9 @@ static void copy_enlightened_to_vmcs12(struct vcpu_vmx *vmx, u32 hv_clean_fields
vmcs12->tpr_threshold = evmcs->tpr_threshold;
vmcs12->guest_rip = evmcs->guest_rip;
+ if ((hv_clean_fields & runtime_controls) != runtime_controls)
+ vmx->nested.force_msr_bitmap_recalc = true;
+
if (unlikely(!(hv_clean_fields &
HV_VMX_ENLIGHTENED_CLEAN_FIELD_ENLIGHTENMENTSCONTROL))) {
hv_vcpu->nested.pa_page_gpa = evmcs->partition_assist_page;
diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c
index 6c842e9191a5..00850a3a77d6 100644
--- a/arch/x86/kvm/vmx/tdx.c
+++ b/arch/x86/kvm/vmx/tdx.c
@@ -2728,11 +2728,6 @@ static tdx_vm_state_guard_t tdx_acquire_vm_state_locks(struct kvm *kvm)
mutex_lock(&kvm->lock);
- if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus)) {
- r = -EBUSY;
- goto out_err;
- }
-
r = kvm_lock_all_vcpus(kvm);
if (r)
goto out_err;
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index cf7fe835c4ad..c844def6dd95 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -791,7 +791,6 @@ struct kvm {
/* The current active memslot set for each address space */
struct kvm_memslots __rcu *memslots[KVM_MAX_NR_ADDRESS_SPACES];
struct xarray vcpu_array;
- DECLARE_BITMAP(vcpu_ids, KVM_MAX_VCPU_IDS);
/*
* Protected by slots_lock, but can be read outside if an
* incorrect answer is acceptable.
diff --git a/tools/testing/selftests/kvm/Makefile.kvm b/tools/testing/selftests/kvm/Makefile.kvm
index 6a1482e3a286..f43549a56f0b 100644
--- a/tools/testing/selftests/kvm/Makefile.kvm
+++ b/tools/testing/selftests/kvm/Makefile.kvm
@@ -103,6 +103,7 @@ TEST_GEN_PROGS_x86 += x86/nested_tdp_fault_test
TEST_GEN_PROGS_x86 += x86/nested_tsc_adjust_test
TEST_GEN_PROGS_x86 += x86/nested_tsc_scaling_test
TEST_GEN_PROGS_x86 += x86/nested_vmsave_vmload_test
+TEST_GEN_PROGS_x86 += x86/nested_x2apic_test
TEST_GEN_PROGS_x86 += x86/platform_info_test
TEST_GEN_PROGS_x86 += x86/pmu_counters_test
TEST_GEN_PROGS_x86 += x86/pmu_event_filter_test
@@ -195,6 +196,7 @@ TEST_GEN_PROGS_arm64 += arm64/vgic_v5
TEST_GEN_PROGS_arm64 += arm64/vpmu_counter_access
TEST_GEN_PROGS_arm64 += arm64/no-vgic
TEST_GEN_PROGS_arm64 += arm64/idreg-idst
+TEST_GEN_PROGS_arm64 += arm64/hidden_features
TEST_GEN_PROGS_arm64 += arm64/kvm-uuid
TEST_GEN_PROGS_arm64 += access_tracking_perf_test
TEST_GEN_PROGS_arm64 += arch_timer
diff --git a/tools/testing/selftests/kvm/arm64/hidden_features.c b/tools/testing/selftests/kvm/arm64/hidden_features.c
new file mode 100644
index 000000000000..194d746e7605
--- /dev/null
+++ b/tools/testing/selftests/kvm/arm64/hidden_features.c
@@ -0,0 +1,184 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * hidden_features - Check that a feature's instruction runs in the guest when
+ * its ID register field is advertised, and is UNDEFINED when userspace clears
+ * the field.
+ *
+ * Copyright (c) 2026 Google LLC
+ * Author: Fuad Tabba <fuad.tabba@linux.dev>
+ */
+#include "kvm_util.h"
+#include "processor.h"
+#include "test_util.h"
+
+static volatile bool undef;
+
+static void guest_tlbi_os(void)
+{
+ /* tlbi vmalle1os */
+ asm volatile("sys #0, c8, c1, #0\n\tdsb ish\n\tisb" ::: "memory");
+}
+
+static void guest_mops(void)
+{
+ register u64 *d asm("x0");
+ register u64 n asm("x1");
+ register u64 s asm("x2");
+ u64 buf[8];
+
+ d = buf;
+ n = sizeof(buf);
+ s = 0;
+ /* setp [x0]!, x1!, x2; setm; sete */
+ asm volatile(".inst 0x19c20420\n\t.inst 0x19c24420\n\t.inst 0x19c28420"
+ : "+r"(d), "+r"(n) : "r"(s) : "cc", "memory");
+}
+
+static void guest_tcr2(void)
+{
+ read_sysreg_s(SYS_TCR2_EL1);
+}
+
+static void guest_fpmr(void)
+{
+ read_sysreg_s(SYS_FPMR);
+}
+
+struct feature {
+ const char *name;
+ u64 id_reg;
+ u64 mask;
+ u8 shift;
+ u64 min;
+ void (*insn)(void);
+ bool (*trappable)(struct kvm_vcpu *vcpu);
+};
+
+/* Without FGT, KVM traps a hidden TLBI OS only through HCR_EL2.TTLBOS (FEAT_EVT2). */
+static bool tlbi_os_trappable(struct kvm_vcpu *vcpu)
+{
+ u64 mmfr0 = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(SYS_ID_AA64MMFR0_EL1));
+ u64 mmfr2 = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(SYS_ID_AA64MMFR2_EL1));
+
+ return SYS_FIELD_GET(ID_AA64MMFR0_EL1, FGT, mmfr0) >= ID_AA64MMFR0_EL1_FGT_IMP ||
+ SYS_FIELD_GET(ID_AA64MMFR2_EL1, EVT, mmfr2) >= ID_AA64MMFR2_EL1_EVT_TTLBxS;
+}
+
+#define FEATURE(n, reg, field, min_val, fn, trap) \
+{ \
+ .name = n, \
+ .id_reg = SYS_##reg, \
+ .mask = reg##_##field##_MASK, \
+ .shift = reg##_##field##_SHIFT, \
+ .min = reg##_##field##_##min_val, \
+ .insn = fn, \
+ .trappable = trap, \
+}
+
+static const struct feature features[] = {
+ FEATURE("TLBI OS", ID_AA64ISAR0_EL1, TLB, OS, guest_tlbi_os, tlbi_os_trappable),
+ FEATURE("MOPS", ID_AA64ISAR2_EL1, MOPS, IMP, guest_mops, NULL),
+ FEATURE("TCR2_EL1", ID_AA64MMFR3_EL1, TCRX, IMP, guest_tcr2, NULL),
+ FEATURE("FPMR", ID_AA64PFR2_EL1, FPMR, IMP, guest_fpmr, NULL),
+};
+
+static void guest_code(const struct feature *feat)
+{
+ undef = false;
+ feat->insn();
+ GUEST_SYNC(undef);
+ GUEST_DONE();
+}
+
+static void guest_undef_handler(struct ex_regs *regs)
+{
+ undef = true;
+ regs->pc += 4;
+}
+
+static bool run(const struct feature *feat, bool hide)
+{
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
+ bool got = false;
+ u64 val;
+
+ vm = vm_create_with_one_vcpu(&vcpu, (void *)guest_code);
+ vm_init_descriptor_tables(vm);
+ vcpu_init_descriptor_tables(vcpu);
+ vm_install_sync_handler(vm, VECTOR_SYNC_CURRENT, ESR_ELx_EC_UNKNOWN, guest_undef_handler);
+ vcpu_args_set(vcpu, 1, feat);
+
+ if (hide) {
+ val = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg));
+ vcpu_set_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg), val & ~feat->mask);
+ }
+
+ for (;;) {
+ vcpu_run(vcpu);
+ switch (get_ucall(vcpu, &uc)) {
+ case UCALL_SYNC:
+ got = uc.args[1];
+ break;
+ case UCALL_ABORT:
+ REPORT_GUEST_ASSERT(uc);
+ break;
+ case UCALL_DONE:
+ kvm_vm_free(vm);
+ return got;
+ default:
+ TEST_FAIL("Unknown ucall %lu", uc.cmd);
+ }
+ }
+}
+
+static void probe_feature(const struct feature *feat, bool *present, bool *trappable)
+{
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ u64 val;
+
+ vm = vm_create_with_one_vcpu(&vcpu, NULL);
+ val = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg));
+ *present = ((val & feat->mask) >> feat->shift) >= feat->min;
+ *trappable = !feat->trappable || feat->trappable(vcpu);
+ kvm_vm_free(vm);
+}
+
+int main(void)
+{
+ const struct feature *feat;
+ bool present, trappable;
+ int i;
+
+ test_disable_default_vgic();
+
+ ksft_print_header();
+ ksft_set_plan(ARRAY_SIZE(features) * 2);
+
+ for (i = 0; i < ARRAY_SIZE(features); i++) {
+ feat = &features[i];
+
+ probe_feature(feat, &present, &trappable);
+ if (!present) {
+ ksft_test_result_skip("%s advertised, not supported\n", feat->name);
+ ksft_test_result_skip("%s hidden, not supported\n", feat->name);
+ continue;
+ }
+
+ if (run(feat, false))
+ ksft_test_result_fail("%s advertised, UNDEF\n", feat->name);
+ else
+ ksft_test_result_pass("%s advertised\n", feat->name);
+
+ if (!trappable)
+ ksft_test_result_skip("%s hidden, not trappable\n", feat->name);
+ else if (run(feat, true))
+ ksft_test_result_pass("%s hidden\n", feat->name);
+ else
+ ksft_test_result_fail("%s hidden, no UNDEF\n", feat->name);
+ }
+
+ ksft_finished();
+}
diff --git a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c
new file mode 100644
index 000000000000..ce204ce29a9c
--- /dev/null
+++ b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c
@@ -0,0 +1,222 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include "test_util.h"
+#include "kvm_util.h"
+#include "processor.h"
+#include "vmx.h"
+#include "svm_util.h"
+
+/*
+ * Use the kernel's posted interrupt vectors to minimize the risk of crashing
+ * the host if KVM is buggy. Note, the vectors aren't set in stone, ideally
+ * these will be kept up-to-date if the kernel vectors change, but it's "fine"
+ * if they are stale.
+ */
+#define POSTED_INTR_VECTOR 0xf2
+#define POSTED_INTR_WAKEUP_VECTOR 0xf1
+#define POSTED_INTR_NESTED_VECTOR 0xf0
+
+static bool inhibit_apicv;
+static volatile unsigned int nr_irqs;
+
+static void guest_irq_handler(struct ex_regs *regs)
+{
+ nr_irqs++;
+ x2apic_write_reg(APIC_EOI, 0);
+}
+
+static void l2_guest_code(void)
+{
+ if (inhibit_apicv)
+ wrmsr(MSR_IA32_APICBASE, rdmsr(MSR_IA32_APICBASE) & GENMASK_ULL(11, 0));
+
+ for (;;) {
+ x2apic_write_reg(APIC_TASKPRI, 0xf0);
+ GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0);
+
+ asm volatile("cpuid" ::: "eax", "ebx", "ecx", "edx");
+ }
+}
+
+static void l1_svm_code(struct svm_test_data *svm)
+{
+ struct vmcb_control_area *ctrl = &svm->vmcb->control;
+
+ generic_svm_setup(svm, l2_guest_code);
+ ctrl->intercept |= BIT_ULL(INTERCEPT_CPUID) | BIT_ULL(INTERCEPT_MSR_PROT);
+
+ run_guest(svm->vmcb, svm->vmcb_gpa);
+ GUEST_ASSERT_EQ(ctrl->exit_code, SVM_EXIT_CPUID);
+
+ stgi();
+ x2apic_write_reg(APIC_TASKPRI, 0);
+}
+
+static void l1_vmx_code(struct vmx_pages *vmx, struct hyperv_test_pages *hv_pages)
+{
+ u64 control;
+
+ if (hv_pages) {
+ wrmsr(HV_X64_MSR_GUEST_OS_ID, HYPERV_LINUX_OS_ID);
+ enable_vp_assist(hv_pages->vp_assist_gpa, hv_pages->vp_assist);
+ evmcs_enable();
+ }
+
+ GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true);
+
+ if (hv_pages) {
+ GUEST_ASSERT(load_evmcs(hv_pages));
+ current_evmcs->hv_enlightenments_control.msr_bitmap = 1;
+ } else {
+ GUEST_ASSERT(load_vmcs(vmx));
+ }
+
+ prepare_vmcs(vmx, NULL);
+ GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, (unsigned long)l2_guest_code), 0);
+
+ control = vmreadz(PIN_BASED_VM_EXEC_CONTROL);
+ control |= PIN_BASED_EXT_INTR_MASK;
+ vmwrite(PIN_BASED_VM_EXEC_CONTROL, control);
+
+ control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
+ control |= CPU_BASED_USE_MSR_BITMAPS | CPU_BASED_TPR_SHADOW;
+ GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0);
+
+ control = vmreadz(SECONDARY_VM_EXEC_CONTROL);
+ control |= SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE |
+ SECONDARY_EXEC_APIC_REGISTER_VIRT |
+ SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY;
+ control &= (rdmsr(MSR_IA32_VMX_PROCBASED_CTLS2) >> 32);
+ GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0);
+
+ GUEST_ASSERT(!vmlaunch());
+ GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID);
+ GUEST_ASSERT_EQ(vmwrite(GUEST_RIP,
+ vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0);
+}
+
+static void l1_vmx_code_part2(void)
+{
+ u64 control;
+
+ control = vmreadz(CPU_BASED_VM_EXEC_CONTROL);
+ control &= ~CPU_BASED_TPR_SHADOW;
+ GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0);
+
+ control = vmreadz(SECONDARY_VM_EXEC_CONTROL);
+ control &= ~(SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE |
+ SECONDARY_EXEC_APIC_REGISTER_VIRT |
+ SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY);
+ GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0);
+
+ GUEST_ASSERT(!vmresume());
+ GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID);
+ GUEST_ASSERT_EQ(vmwrite(GUEST_RIP,
+ vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0);
+}
+
+static void l1_test_x2apic_intercepts(void)
+{
+ GUEST_ASSERT_EQ(nr_irqs, 0);
+
+ sti_nop();
+
+ GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0);
+
+ x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_VECTOR);
+ GUEST_ASSERT_EQ(nr_irqs, 0);
+
+ x2apic_write_reg(APIC_TASKPRI, 0xff);
+ GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xff);
+ x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_WAKEUP_VECTOR);
+ GUEST_ASSERT_EQ(nr_irqs, 0);
+
+ x2apic_write_reg(APIC_TASKPRI, 0);
+ GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0);
+
+ x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_NESTED_VECTOR);
+ GUEST_ASSERT_EQ(nr_irqs, 3);
+
+ nr_irqs = 0;
+}
+
+static void l1_guest_code(void *test_data, void *hv_pages)
+{
+ x2apic_enable();
+
+ if (this_cpu_has(X86_FEATURE_SVM))
+ l1_svm_code(test_data);
+ else
+ l1_vmx_code(test_data, hv_pages);
+
+ GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0);
+ x2apic_write_reg(APIC_TASKPRI, 0xf0);
+
+ l1_test_x2apic_intercepts();
+
+ if (this_cpu_has(X86_FEATURE_VMX)) {
+ l1_vmx_code_part2();
+ l1_test_x2apic_intercepts();
+ }
+
+ GUEST_DONE();
+}
+
+static void test_x2apic_intercepts(bool with_inhibit_apicv, bool use_evmcs)
+{
+ gva_t nested_test_data_gva, hv_pages_gva = 0;
+ struct kvm_vcpu *vcpu;
+ struct kvm_vm *vm;
+ struct ucall uc;
+
+ inhibit_apicv = with_inhibit_apicv;
+
+ vm = vm_create_with_one_vcpu(&vcpu, l1_guest_code);
+ vm_install_exception_handler(vm, POSTED_INTR_VECTOR, guest_irq_handler);
+ vm_install_exception_handler(vm, POSTED_INTR_WAKEUP_VECTOR, guest_irq_handler);
+ vm_install_exception_handler(vm, POSTED_INTR_NESTED_VECTOR, guest_irq_handler);
+
+ sync_global_to_guest(vm, inhibit_apicv);
+
+ if (kvm_cpu_has(X86_FEATURE_SVM))
+ vcpu_alloc_svm(vm, &nested_test_data_gva);
+ else
+ vcpu_alloc_vmx(vm, &nested_test_data_gva);
+
+ if (use_evmcs) {
+ vcpu_set_hv_cpuid(vcpu);
+ vcpu_enable_evmcs(vcpu);
+
+ vcpu_alloc_hyperv_test_pages(vm, &hv_pages_gva);
+ }
+
+ vcpu_args_set(vcpu, 2, nested_test_data_gva, hv_pages_gva);
+
+ vcpu_run(vcpu);
+
+ TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO);
+
+ switch (get_ucall(vcpu, &uc)) {
+ case UCALL_DONE:
+ break;
+ case UCALL_ABORT:
+ REPORT_GUEST_ASSERT(uc);
+ break;
+ default:
+ TEST_FAIL("Expected DONE, got unexpected ucall %lu", uc.cmd);
+ }
+
+ kvm_vm_free(vm);
+}
+
+int main(int argc, char *argv[])
+{
+ TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_SVM) || kvm_cpu_has(X86_FEATURE_VMX));
+
+ test_x2apic_intercepts(true, false);
+ test_x2apic_intercepts(false, false);
+
+ if (kvm_has_cap(KVM_CAP_HYPERV_ENLIGHTENED_VMCS)) {
+ test_x2apic_intercepts(true, true);
+ test_x2apic_intercepts(false, true);
+ }
+}
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 85f42289748d..5a782849521c 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -1363,6 +1363,9 @@ int kvm_trylock_all_vcpus(struct kvm *kvm)
lockdep_assert_held(&kvm->lock);
+ if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm)))
+ return -EBUSY;
+
kvm_for_each_vcpu(i, vcpu, kvm)
if (!mutex_trylock_nest_lock(&vcpu->mutex, &kvm->lock))
goto out_unlock;
@@ -1386,6 +1389,9 @@ int kvm_lock_all_vcpus(struct kvm *kvm)
lockdep_assert_held(&kvm->lock);
+ if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm)))
+ return -EBUSY;
+
kvm_for_each_vcpu(i, vcpu, kvm) {
r = mutex_lock_killable_nest_lock(&vcpu->mutex, &kvm->lock);
if (r)
@@ -4182,6 +4188,8 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
struct kvm_vcpu *vcpu;
struct page *page;
+ guard(mutex)(&kvm->lock);
+
/*
* KVM tracks vCPU IDs as 'int', be kind to userspace and reject
* too-large values instead of silently truncating.
@@ -4194,26 +4202,17 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
if (id >= KVM_MAX_VCPU_IDS)
return -EINVAL;
- mutex_lock(&kvm->lock);
- if (kvm->created_vcpus >= kvm->max_vcpus) {
- mutex_unlock(&kvm->lock);
+ if (kvm->created_vcpus >= kvm->max_vcpus)
return -EINVAL;
- }
- if (test_bit(id, kvm->vcpu_ids)) {
- mutex_unlock(&kvm->lock);
+ if (kvm_get_vcpu_by_id(kvm, id))
return -EEXIST;
- }
r = kvm_arch_vcpu_precreate(kvm, id);
- if (r) {
- mutex_unlock(&kvm->lock);
+ if (r)
return r;
- }
kvm->created_vcpus++;
- __set_bit(id, kvm->vcpu_ids);
- mutex_unlock(&kvm->lock);
vcpu = kmem_cache_zalloc(kvm_vcpu_cache, GFP_KERNEL_ACCOUNT);
if (!vcpu) {
@@ -4244,13 +4243,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
goto arch_vcpu_destroy;
}
- mutex_lock(&kvm->lock);
-
- if (WARN_ON_ONCE(kvm_get_vcpu_by_id(kvm, id))) {
- r = -EEXIST;
- goto unlock_vcpu_destroy;
- }
-
/*
* Set the vCPU's index *before* the vCPU is reachable by other tasks.
* Unwind the index back to -1 on failure so that KVM can use the index
@@ -4284,7 +4276,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
atomic_inc(&kvm->online_vcpus);
mutex_unlock(&vcpu->mutex);
- mutex_unlock(&kvm->lock);
kvm_arch_vcpu_postcreate(vcpu);
kvm_create_vcpu_debugfs(vcpu);
return r;
@@ -4295,7 +4286,6 @@ kvm_put_xa_erase:
xa_erase(&kvm->vcpu_array, vcpu->vcpu_idx);
unlock_vcpu_destroy:
vcpu->vcpu_idx = -1;
- mutex_unlock(&kvm->lock);
kvm_dirty_ring_free(&vcpu->dirty_ring);
arch_vcpu_destroy:
kvm_arch_vcpu_destroy(vcpu);
@@ -4304,10 +4294,7 @@ vcpu_free_run_page:
vcpu_free:
kmem_cache_free(kvm_vcpu_cache, vcpu);
vcpu_decrement:
- mutex_lock(&kvm->lock);
kvm->created_vcpus--;
- __clear_bit(id, kvm->vcpu_ids);
- mutex_unlock(&kvm->lock);
return r;
}