diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-01 10:57:19 -0700 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-01 10:57:19 -0700 |
| commit | a940b03cee1524c10c16e0f73ec878bbd36202a2 (patch) | |
| tree | 931c568ee3aa3bbab3da762e80adc54c4f4bce0a | |
| parent | d24e8ac715de2e16a53c144005b1863660a5fbea (diff) | |
| parent | f9bfc323e76120c2cd1fdea93bbf315eec5eb934 (diff) | |
Merge tag 'for-linus' of git://git.kernel.org/pub/scm/virt/kvm/kvm
Pull kvm fixes from Paolo Bonzini:
"The most intrusive change is reverting a commit from 7.3-rc1 that made
struct kvm a bit too large, and fixing the same issue otherwise.
There are again a lot of selftests lines; the sheer number of commits
is not small but I don't expect much more for 7.3 due to people
travelling to Plumbers next week.
ARM:
- Take a reference on the last IRQ loaded into an LR to prevent it
from being freed while running the guest (Marc Zyngier)
- Ensure that the ITS MOVALL command only affects LPIs that were
previously affined to the source redistributor (Marc Zyngier)
- Fix + test for honoring the host's trap configuration when running
non-protected VMs while KVM is in protected mode (Fuad Tabba)
- Use the host stage-1 mapping granularity for VM_PFNMAP mappings at
stage-2 (Mostafa Saleh)
x86:
Various bugfixes where the guest could do stupid things on purpose to
cause problems in the host:
- Failed VMRUNs can cause pending TLB flushes to be dropped, and in
general some actions done through VMCB control fields have to be
redone if VMRUN fails
- Toggling MSR interceptions or eVMCS execution controls can cause
the host to use a stale MSR permission bitmap
- Bad page tables can cause a WARN.
Also fix issues in last week's pull request (my fault, for changing
email workflow and thus missing feedback sent to kvm@ but not LKML).
Generic:
- Take kvm_lock when creating vCPUs. For almost two decades everybody
thought it was not done for some unspecified performance reasons,
but in reality it was only done because kvm_lock was originally a
spinlock.
This is a better fix than 97d65b544f48 ("KVM: Check for duplicate
vcpu_id as early as possible", from the 7.3 merge window), and does
not waste 2K per VM, hence its inclusion here"
* tag 'for-linus' of git://git.kernel.org/pub/scm/virt/kvm/kvm: (29 commits)
KVM: arm64: Use stage-1 leaf size for VM_PFNMAP
KVM: arm64: selftests: Check a feature hidden in an ID register is UNDEF
KVM: arm64: Use the host's HCR_EL2 for non-protected VMs in pKVM
KVM: arm64: Clear HCR_EL2.RW for 32-bit non-protected vCPUs
KVM: arm64: Apply the fine-grained UNDEFs without FEAT_FGT
KVM: arm64: vgic-its: Fix MOVALL handling of source redistributor
KVM: arm64: vgic: Take a refcount on IRQs referenced by last_lr_irq
KVM: arm64: vgic: Allow last_lr_irq to be NULL when LRs are not overflowing
KVM: SEV: Do cache maintenance on the source VM *before* clearing SEV state
KVM: SEV: Nullify "have run CPUs" mask pointer when freeing it
KVM: selftests: Extend nested x2APIC test to validate using eVMCS for vmcs12
KVM: selftests: Extend nested x2APIC test to validate disabling x2APIC virt
KVM: selftests: Verify that L0's TPR doesn't get clobbered
KVM: selftests: Run the nested x2APIC with and without APICv being inhibited in L2
KVM: selftests: Add x2APIC MSR test for inhibiting APICv while nested
KVM: nVMX: Force MSR bitmap refresh if runtime eVMCS controls are modified
KVM: SVM: Use the active VMCB's MSR bitmap when checking if MSR is intercepted
KVM: SVM: Sync guest's PERF_CNTR_GLOBAL_CTL from h/w only on successful VMRUN
KVM: SVM: Don't mark ASID fields as dirty when setting control.tlb_ctl
KVM: SVM: Update control fields on #VMEXIT if and only if VMRUN succeeded
...
| -rw-r--r-- | arch/arm64/kvm/emulate-nested.c | 3 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/include/nvhe/pkvm.h | 9 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/hyp-main.c | 7 | ||||
| -rw-r--r-- | arch/arm64/kvm/hyp/nvhe/pkvm.c | 25 | ||||
| -rw-r--r-- | arch/arm64/kvm/mmu.c | 47 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic-init.c | 12 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic-its.c | 12 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic-v2.c | 6 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic-v3.c | 6 | ||||
| -rw-r--r-- | arch/arm64/kvm/vgic/vgic.c | 21 | ||||
| -rw-r--r-- | arch/powerpc/kvm/book3s_hv.c | 2 | ||||
| -rw-r--r-- | arch/riscv/kvm/aia_device.c | 2 | ||||
| -rw-r--r-- | arch/s390/kvm/s390/s390.c | 5 | ||||
| -rw-r--r-- | arch/x86/kvm/mmu/mmu.c | 6 | ||||
| -rw-r--r-- | arch/x86/kvm/svm/sev.c | 57 | ||||
| -rw-r--r-- | arch/x86/kvm/svm/svm.c | 43 | ||||
| -rw-r--r-- | arch/x86/kvm/svm/svm.h | 1 | ||||
| -rw-r--r-- | arch/x86/kvm/vmx/nested.c | 6 | ||||
| -rw-r--r-- | arch/x86/kvm/vmx/tdx.c | 5 | ||||
| -rw-r--r-- | include/linux/kvm_host.h | 1 | ||||
| -rw-r--r-- | tools/testing/selftests/kvm/Makefile.kvm | 2 | ||||
| -rw-r--r-- | tools/testing/selftests/kvm/arm64/hidden_features.c | 184 | ||||
| -rw-r--r-- | tools/testing/selftests/kvm/x86/nested_x2apic_test.c | 222 | ||||
| -rw-r--r-- | virt/kvm/kvm_main.c | 35 |
24 files changed, 554 insertions, 165 deletions
diff --git a/arch/arm64/kvm/emulate-nested.c b/arch/arm64/kvm/emulate-nested.c index 3806ff0920fe..f59d89d04eca 100644 --- a/arch/arm64/kvm/emulate-nested.c +++ b/arch/arm64/kvm/emulate-nested.c @@ -2386,9 +2386,6 @@ int __init populate_nv_trap_config(void) print_nv_trap_error(fgt, "FGT bit is reserved", ret); } - if (!cpus_have_final_cap(ARM64_HAS_FGT)) - continue; - prev = xa_store(&sr_forward_xa, enc, xa_mk_value(tc.val), GFP_KERNEL); diff --git a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h index c904647d2f76..5c050f21066a 100644 --- a/arch/arm64/kvm/hyp/include/nvhe/pkvm.h +++ b/arch/arm64/kvm/hyp/include/nvhe/pkvm.h @@ -13,6 +13,15 @@ #include <nvhe/spinlock.h> /* + * HCR_EL2 bits EL2 takes from the host on each entry, per VM type. The rest + * are EL2's own and nothing the host sets there reaches the guest. + */ +#define PKVM_HCR_EL2_HOST_PVM (HCR_EL2_TWI | HCR_EL2_TWE | HCR_EL2_VSE) +#define PKVM_HCR_EL2_HOST_NPVM (PKVM_HCR_EL2_HOST_PVM | HCR_EL2_VI | HCR_EL2_VF | \ + HCR_EL2_TVM | HCR_EL2_TID2 | HCR_EL2_TID4 | \ + HCR_EL2_TID5 | HCR_EL2_TTLBOS) + +/* * Holds the relevant data for maintaining the vcpu state completely at hyp. */ struct pkvm_hyp_vcpu { diff --git a/arch/arm64/kvm/hyp/nvhe/hyp-main.c b/arch/arm64/kvm/hyp/nvhe/hyp-main.c index 9a3b92e626ad..ac64a036b0a9 100644 --- a/arch/arm64/kvm/hyp/nvhe/hyp-main.c +++ b/arch/arm64/kvm/hyp/nvhe/hyp-main.c @@ -216,6 +216,7 @@ static void sync_debug_state(struct pkvm_hyp_vcpu *hyp_vcpu) static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu) { struct kvm_vcpu *host_vcpu = hyp_vcpu->host_vcpu; + u64 host_hcr_mask = PKVM_HCR_EL2_HOST_PVM; fpsimd_sve_flush(); flush_debug_state(hyp_vcpu); @@ -228,6 +229,7 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu) if (!pkvm_hyp_vcpu_is_protected(hyp_vcpu)) { if (vcpu_get_flag(host_vcpu, PKVM_HOST_STATE_DIRTY)) flush_hyp_vcpu_state(hyp_vcpu); + host_hcr_mask = PKVM_HCR_EL2_HOST_NPVM; } else { hyp_vcpu->vcpu.arch.ctxt = host_vcpu->arch.ctxt; } @@ -241,9 +243,8 @@ static void flush_hyp_vcpu(struct pkvm_hyp_vcpu *hyp_vcpu) * trap-control bit, so it must flow to the hyp vCPU alongside TWI/TWE * for the vSError to be delivered. sync_hyp_vcpu() reflects it back. */ - hyp_vcpu->vcpu.arch.hcr_el2 &= ~(HCR_TWI | HCR_TWE | HCR_VSE); - hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) & - (HCR_TWI | HCR_TWE | HCR_VSE); + hyp_vcpu->vcpu.arch.hcr_el2 &= ~host_hcr_mask; + hyp_vcpu->vcpu.arch.hcr_el2 |= READ_ONCE(host_vcpu->arch.hcr_el2) & host_hcr_mask; hyp_vcpu->vcpu.arch.iflags = host_vcpu->arch.iflags; diff --git a/arch/arm64/kvm/hyp/nvhe/pkvm.c b/arch/arm64/kvm/hyp/nvhe/pkvm.c index bb3e0dc0676e..6290c4b62659 100644 --- a/arch/arm64/kvm/hyp/nvhe/pkvm.c +++ b/arch/arm64/kvm/hyp/nvhe/pkvm.c @@ -30,6 +30,7 @@ unsigned int kvm_host_sve_max_vl; */ static DEFINE_PER_CPU(struct pkvm_hyp_vcpu *, loaded_hyp_vcpu); +/* The PKVM_HCR_EL2_HOST_{PVM,NPVM} bits of this value come from the host on each entry. */ static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu) { vcpu->arch.hcr_el2 = HCR_GUEST_FLAGS; @@ -47,18 +48,17 @@ static void pkvm_vcpu_reset_hcr(struct kvm_vcpu *vcpu) if (cpus_have_final_cap(ARM64_HAS_STAGE2_FWB)) vcpu->arch.hcr_el2 |= HCR_FWB; - if (cpus_have_final_cap(ARM64_HAS_EVT) && - !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) && - kvm_read_vm_id_reg(vcpu->kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0)) - vcpu->arch.hcr_el2 |= HCR_TID4; - else - vcpu->arch.hcr_el2 |= HCR_TID2; + /* + * Without AArch32 EL1, leave RW set and let the entry fail with an + * illegal exception return: the *32_EL2 registers EL2 would otherwise + * switch are UNDEFINED there. + */ + if (vcpu_has_feature(vcpu, KVM_ARM_VCPU_EL1_32BIT) && + cpus_have_final_cap(ARM64_HAS_32BIT_EL1)) + vcpu->arch.hcr_el2 &= ~HCR_EL2_RW; if (vcpu_has_ptrauth(vcpu)) vcpu->arch.hcr_el2 |= (HCR_API | HCR_APK); - - if (kvm_has_mte(vcpu->kvm)) - vcpu->arch.hcr_el2 |= HCR_ATA; } static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu) @@ -76,6 +76,13 @@ static void pvm_init_traps_hcr(struct kvm_vcpu *vcpu) */ val |= HCR_TACR | HCR_TIDCP | HCR_TID3 | HCR_TID1; + if (cpus_have_final_cap(ARM64_HAS_EVT) && + !cpus_have_final_cap(ARM64_MISMATCHED_CACHE_TYPE) && + kvm_read_vm_id_reg(kvm, SYS_CTR_EL0) == read_cpuid(CTR_EL0)) + val |= HCR_EL2_TID4; + else + val |= HCR_EL2_TID2; + if (!kvm_has_feat(kvm, ID_AA64PFR0_EL1, RAS, IMP)) { val |= HCR_TERR | HCR_TEA; val &= ~(HCR_FIEN); diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index 2d44cd6a5aed..bf6d6526639f 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -1468,32 +1468,11 @@ transparent_hugepage_adjust(struct kvm *kvm, struct kvm_memory_slot *memslot, return PAGE_SIZE; } -static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva) +static int get_vma_page_shift(struct vm_area_struct *vma) { - unsigned long pa; - - if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP)) + if (is_vm_hugetlb_page(vma)) return huge_page_shift(hstate_vma(vma)); - if (!(vma->vm_flags & VM_PFNMAP)) - return PAGE_SHIFT; - - VM_BUG_ON(is_vm_hugetlb_page(vma)); - - pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start); - -#ifndef __PAGETABLE_PMD_FOLDED - if ((hva & (PUD_SIZE - 1)) == (pa & (PUD_SIZE - 1)) && - ALIGN_DOWN(hva, PUD_SIZE) >= vma->vm_start && - ALIGN(hva, PUD_SIZE) <= vma->vm_end) - return PUD_SHIFT; -#endif - - if ((hva & (PMD_SIZE - 1)) == (pa & (PMD_SIZE - 1)) && - ALIGN_DOWN(hva, PMD_SIZE) >= vma->vm_start && - ALIGN(hva, PMD_SIZE) <= vma->vm_end) - return PMD_SHIFT; - return PAGE_SHIFT; } @@ -1794,7 +1773,7 @@ static short kvm_s2_resolve_vma_size(const struct kvm_s2_fault_desc *s2fd, vma_shift = PAGE_SHIFT; } else { s2vi->max_map_size = PUD_SIZE; - vma_shift = get_vma_page_shift(vma, s2fd->hva); + vma_shift = get_vma_page_shift(vma); } switch (vma_shift) { @@ -1952,16 +1931,6 @@ static int kvm_s2_fault_pin_pfn(const struct kvm_s2_fault_desc *s2fd, return -EFAULT; } } else { - /* - * If the page was identified as device early by looking at - * the VMA flags, vma_pagesize is already representing the - * largest quantity we can map. If instead it was mapped - * via __kvm_faultin_pfn(), vma_pagesize is set to PAGE_SIZE - * and must not be upgraded. - * - * In both cases, we don't let transparent_hugepage_adjust() - * change things at the last minute. - */ s2vi->map_non_cacheable = true; } @@ -2051,10 +2020,10 @@ static int kvm_s2_fault_map(const struct kvm_s2_fault_desc *s2fd, /* * If we are not forced to use page mapping, check if we are - * backed by a THP and thus use block mapping if possible. + * backed by a huge stage-1 mapping and thus use block mapping if + * possible. */ - if (mapping_size == PAGE_SIZE && - !(s2vi->max_map_size == PAGE_SIZE || s2vi->map_non_cacheable)) { + if (mapping_size == PAGE_SIZE && s2vi->max_map_size != PAGE_SIZE) { if (perm_fault_granule > PAGE_SIZE) { mapping_size = perm_fault_granule; } else { @@ -2135,10 +2104,6 @@ static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd) return ret; } - /* - * Let's check if we will get back a huge page backed by hugetlbfs, or - * get block mapping for device MMIO region. - */ ret = kvm_s2_fault_pin_pfn(s2fd, &s2vi); if (ret != 1) return ret; diff --git a/arch/arm64/kvm/vgic/vgic-init.c b/arch/arm64/kvm/vgic/vgic-init.c index 4012df6002ea..a58575df36e9 100644 --- a/arch/arm64/kvm/vgic/vgic-init.c +++ b/arch/arm64/kvm/vgic/vgic-init.c @@ -97,6 +97,9 @@ int kvm_vgic_create(struct kvm *kvm, u32 type) /* * - Acquiring the vCPU mutex for every *online* vCPU to prevent * concurrent vCPU ioctls for vCPUs already visible to userspace. + * This also ensures KVM isn't in the middle of creating a vCPU, + * i.e. that there are no vCPUs that have been created but aren't + * yet fully online. */ ret = -EBUSY; if (kvm_trylock_all_vcpus(kvm)) @@ -105,18 +108,11 @@ int kvm_vgic_create(struct kvm *kvm, u32 type) /* * - Taking the config_lock which protects VGIC data structures such * as the per-vCPU arrays of private IRQs (SGIs, PPIs). - */ - mutex_lock(&kvm->arch.config_lock); - - /* - * - Bailing on the entire thing if a vCPU is in the middle of creation, - * dropped the kvm->lock, but hasn't reached kvm_arch_vcpu_create(). * * The whole combination of this guarantees that no vCPU can get into * KVM with a VGIC configuration inconsistent with the VM's VGIC. */ - if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus)) - goto out_unlock; + mutex_lock(&kvm->arch.config_lock); if (irqchip_in_kernel(kvm)) { ret = -EEXIST; diff --git a/arch/arm64/kvm/vgic/vgic-its.c b/arch/arm64/kvm/vgic/vgic-its.c index 0904ae850c35..c10a11373e10 100644 --- a/arch/arm64/kvm/vgic/vgic-its.c +++ b/arch/arm64/kvm/vgic/vgic-its.c @@ -319,12 +319,16 @@ static int update_lpi_config(struct kvm *kvm, struct vgic_irq *irq, return ret; } -static int update_affinity(struct vgic_irq *irq, struct kvm_vcpu *vcpu) +static int update_affinity(struct vgic_irq *irq, + struct kvm_vcpu *from_vcpu, struct kvm_vcpu *vcpu) { struct its_vlpi_map map; int ret; guard(raw_spinlock_irqsave)(&irq->irq_lock); + if (from_vcpu && irq->target_vcpu != from_vcpu) + return 0; + irq->target_vcpu = vcpu; if (!irq->hw) @@ -362,7 +366,7 @@ static void update_affinity_ite(struct kvm *kvm, struct its_ite *ite) return; vcpu = collection_to_vcpu(kvm, ite->collection); - update_affinity(ite->irq, vcpu); + update_affinity(ite->irq, NULL, vcpu); } /* @@ -856,7 +860,7 @@ static int vgic_its_cmd_handle_movi(struct kvm *kvm, struct vgic_its *its, vgic_its_invalidate_cache(its); - return update_affinity(ite->irq, vcpu); + return update_affinity(ite->irq, NULL, vcpu); } static bool __is_visible_gfn_locked(struct vgic_its *its, gpa_t gpa) @@ -1383,7 +1387,7 @@ static int vgic_its_cmd_handle_movall(struct kvm *kvm, struct vgic_its *its, if (!irq) continue; - update_affinity(irq, vcpu2); + update_affinity(irq, vcpu1, vcpu2); vgic_put_irq(kvm, irq); } diff --git a/arch/arm64/kvm/vgic/vgic-v2.c b/arch/arm64/kvm/vgic/vgic-v2.c index 7182f63fc938..70cc53ba810a 100644 --- a/arch/arm64/kvm/vgic/vgic-v2.c +++ b/arch/arm64/kvm/vgic/vgic-v2.c @@ -122,6 +122,10 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu) for (int lr = 0; lr < vgic_cpu->vgic_v2.used_lrs; lr++) vgic_v2_fold_lr(vcpu, cpuif->vgic_lr[lr]); + cpuif->used_lrs = 0; + if (!irq) + return; + /* See the GICv3 equivalent for the EOIcount handling rationale */ list_for_each_entry_continue(irq, &vgic_cpu->ap_list_head, ap_list) { u32 lr; @@ -144,8 +148,6 @@ void vgic_v2_fold_lr_state(struct kvm_vcpu *vcpu) vgic_v2_fold_lr(vcpu, lr); eoicount--; } - - cpuif->used_lrs = 0; } void vgic_v2_deactivate(struct kvm_vcpu *vcpu, u32 val) diff --git a/arch/arm64/kvm/vgic/vgic-v3.c b/arch/arm64/kvm/vgic/vgic-v3.c index 726e20a1da6e..346bacb3198f 100644 --- a/arch/arm64/kvm/vgic/vgic-v3.c +++ b/arch/arm64/kvm/vgic/vgic-v3.c @@ -155,6 +155,10 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu) for (int lr = 0; lr < cpuif->used_lrs; lr++) vgic_v3_fold_lr(vcpu, cpuif->vgic_lr[lr]); + cpuif->used_lrs = 0; + if (!irq) + return; + /* * EOIMode=0: use EOIcount to emulate deactivation. We are * guaranteed to deactivate in reverse order of the activation, so @@ -188,8 +192,6 @@ void vgic_v3_fold_lr_state(struct kvm_vcpu *vcpu) vgic_v3_fold_lr(vcpu, lr); eoicount--; } - - cpuif->used_lrs = 0; } void vgic_v3_deactivate(struct kvm_vcpu *vcpu, u64 val) diff --git a/arch/arm64/kvm/vgic/vgic.c b/arch/arm64/kvm/vgic/vgic.c index b25303d9919f..5cf5a1ef86cd 100644 --- a/arch/arm64/kvm/vgic/vgic.c +++ b/arch/arm64/kvm/vgic/vgic.c @@ -853,6 +853,17 @@ retry: goto retry; } + /* + * Fix the last_lr_irq refcount which was obtained while + * populating the LRs. This can also result in the LPI being + * deleted. + */ + irq = *host_data_ptr(last_lr_irq); + if (irq) { + deleted_lpis |= vgic_put_irq_norelease(vcpu->kvm, irq); + *host_data_ptr(last_lr_irq) = NULL; + } + raw_spin_unlock(&vgic_cpu->ap_list_lock); if (unlikely(deleted_lpis)) @@ -866,9 +877,6 @@ static void vgic_fold_state(struct kvm_vcpu *vcpu) return; } - if (!*host_data_ptr(last_lr_irq)) - return; - if (kvm_vgic_global_state.type == VGIC_V2) vgic_v2_fold_lr_state(vcpu); else @@ -1021,11 +1029,14 @@ static void vgic_flush_lr_state(struct kvm_vcpu *vcpu) scoped_guard(raw_spinlock, &irq->irq_lock) { if (likely(vgic_target_oracle(irq) == vcpu)) { vgic_populate_lr(vcpu, irq, count++); - *host_data_ptr(last_lr_irq) = irq; + if (count == kvm_vgic_global_state.nr_lr) { + vgic_get_irq_ref(irq); + *host_data_ptr(last_lr_irq) = irq; + } } } - if (count == kvm_vgic_global_state.nr_lr) + if (*host_data_ptr(last_lr_irq)) break; } diff --git a/arch/powerpc/kvm/book3s_hv.c b/arch/powerpc/kvm/book3s_hv.c index dbac3573b2c8..de05d721edc4 100644 --- a/arch/powerpc/kvm/book3s_hv.c +++ b/arch/powerpc/kvm/book3s_hv.c @@ -3058,7 +3058,6 @@ static int kvmppc_core_vcpu_create_hv(struct kvm_vcpu *vcpu) init_waitqueue_head(&vcpu->arch.cpu_run); - mutex_lock(&kvm->lock); vcore = NULL; err = -EINVAL; if (cpu_has_feature(CPU_FTR_ARCH_300)) { @@ -3091,7 +3090,6 @@ static int kvmppc_core_vcpu_create_hv(struct kvm_vcpu *vcpu) mutex_unlock(&kvm->arch.mmu_setup_lock); } } - mutex_unlock(&kvm->lock); if (!vcore) return err; diff --git a/arch/riscv/kvm/aia_device.c b/arch/riscv/kvm/aia_device.c index efc7c0bcfba9..97833a04268a 100644 --- a/arch/riscv/kvm/aia_device.c +++ b/arch/riscv/kvm/aia_device.c @@ -237,7 +237,7 @@ static int aia_init(struct kvm *kvm) return -EBUSY; /* We might be in the middle of creating a VCPU? */ - if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus)) + if (kvm_is_vcpu_creation_in_progress(kvm)) return -EBUSY; /* Number of sources should be less than or equals number of IDs */ diff --git a/arch/s390/kvm/s390/s390.c b/arch/s390/kvm/s390/s390.c index efd77b9cd36c..e4912db67adc 100644 --- a/arch/s390/kvm/s390/s390.c +++ b/arch/s390/kvm/s390/s390.c @@ -3579,12 +3579,11 @@ void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu) void kvm_arch_vcpu_postcreate(struct kvm_vcpu *vcpu) { - mutex_lock(&vcpu->kvm->lock); preempt_disable(); vcpu->arch.sie_block->epoch = vcpu->kvm->arch.epoch; vcpu->arch.sie_block->epdx = vcpu->kvm->arch.epdx; preempt_enable(); - mutex_unlock(&vcpu->kvm->lock); + if (!kvm_is_ucontrol(vcpu->kvm)) { vcpu->arch.gmap = vcpu->kvm->arch.gmap; sca_add_vcpu(vcpu); @@ -3757,13 +3756,11 @@ static int kvm_s390_vcpu_setup(struct kvm_vcpu *vcpu) kvm_s390_vcpu_pci_setup(vcpu); - mutex_lock(&vcpu->kvm->lock); if (kvm_s390_pv_is_protected(vcpu->kvm)) { rc = kvm_s390_pv_create_cpu(vcpu, &uvrc, &uvrrc); if (rc) kvm_s390_vcpu_unsetup_cmma(vcpu); } - mutex_unlock(&vcpu->kvm->lock); return rc; } diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index 8e62476e477b..a0d608e3fceb 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -2512,6 +2512,12 @@ static void shadow_walk_init_using_root(struct kvm_shadow_walk_iterator *iterato iterator->addr = addr; iterator->shadow_addr = root; + + if (WARN_ON_ONCE(!VALID_PAGE(root)) || kvm_mmu_is_dummy_root(root)) { + iterator->level = 0; + return; + } + iterator->level = vcpu->arch.mmu->root_role.level; if (iterator->level >= PT64_ROOT_4LEVEL && diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index 63eb2155a774..0c1ebb16cec6 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -488,6 +488,20 @@ static void snp_guest_req_cleanup(struct kvm *kvm) sev->guest_resp_buf = NULL; } +static int sev_alloc_have_run_cpus(struct kvm_sev_info *sev) +{ + if (!zalloc_cpumask_var(&sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) + return -ENOMEM; + + return 0; +} + +static void sev_free_have_run_cpus(struct kvm_sev_info *sev) +{ + free_cpumask_var(sev->have_run_cpus); + memset(&sev->have_run_cpus, 0, sizeof(sev->have_run_cpus)); +} + static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp, struct kvm_sev_init *data, unsigned long vm_type) @@ -545,10 +559,9 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp, if (ret) goto e_free_asid; - if (!zalloc_cpumask_var(&sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) { - ret = -ENOMEM; + ret = sev_alloc_have_run_cpus(sev); + if (ret) goto e_free_asid; - } /* This needs to happen after SEV/SNP firmware initialization. */ if (snp_active) { @@ -566,7 +579,7 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm_sev_cmd *argp, return 0; e_free: - free_cpumask_var(sev->have_run_cpus); + sev_free_have_run_cpus(sev); e_free_asid: argp->error = init_args.error; sev_asid_free(sev); @@ -1125,9 +1138,6 @@ static int sev_launch_update_vmsa(struct kvm *kvm, struct kvm_sev_cmd *argp) if (!sev_es_guest(kvm)) return -ENOTTY; - if (kvm_is_vcpu_creation_in_progress(kvm)) - return -EBUSY; - ret = kvm_lock_all_vcpus(kvm); if (ret) return ret; @@ -2035,6 +2045,13 @@ static void sev_migrate_from(struct kvm *dst_kvm, struct kvm *src_kvm) struct kvm_sev_info *mirror; unsigned long i; + /* + * Do cache maintenance on the source VM *before* clearing "SEV active", + * as memory reclaim flows won't trigger cache maintenance on the VM + * once it's no longer an SEV VM. + */ + sev_writeback_caches(src_kvm); + dst->active = true; dst->asid = src->asid; dst->handle = src->handle; @@ -2048,12 +2065,6 @@ static void sev_migrate_from(struct kvm *dst_kvm, struct kvm *src_kvm) src->pages_locked = 0; src->es_active = false; - /* - * Do cache maintenance on the source VM as it is no longer an SEV VM, - * i.e. memory reclaim flows won't trigger cache maintenance on the VM. - */ - sev_writeback_caches(src_kvm); - list_cut_before(&dst->regions_list, &src->regions_list, &src->regions_list); mutex_lock(&sev_mirror_lock); @@ -2121,10 +2132,6 @@ static int sev_check_source_vcpus(struct kvm *dst, struct kvm *src) struct kvm_vcpu *src_vcpu; unsigned long i; - if (kvm_is_vcpu_creation_in_progress(src) || - kvm_is_vcpu_creation_in_progress(dst)) - return -EBUSY; - if (!sev_es_guest(src)) return 0; @@ -2198,10 +2205,9 @@ int sev_vm_move_enc_context_from(struct kvm *kvm, unsigned int source_fd) * does not, i.e. KVM could skip flushes if memory is reclaimed from * the old VM but not the new VM. */ - if (!zalloc_cpumask_var(&dst_sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) { - ret = -ENOMEM; + ret = sev_alloc_have_run_cpus(dst_sev); + if (ret) goto out_source_vcpu; - } sev_migrate_from(kvm, source_kvm); kvm_vm_dead(source_kvm); @@ -2520,9 +2526,6 @@ static int snp_launch_update_vmsa(struct kvm *kvm, struct kvm_sev_cmd *argp) unsigned long i; int ret; - if (kvm_is_vcpu_creation_in_progress(kvm)) - return -EBUSY; - ret = kvm_lock_all_vcpus(kvm); if (ret) return ret; @@ -2898,10 +2901,9 @@ int sev_vm_copy_enc_context_from(struct kvm *kvm, unsigned int source_fd) } mirror_sev = to_kvm_sev_info(kvm); - if (!zalloc_cpumask_var(&mirror_sev->have_run_cpus, GFP_KERNEL_ACCOUNT)) { - ret = -ENOMEM; + ret = sev_alloc_have_run_cpus(mirror_sev); + if (ret) goto e_unlock; - } /* * The mirror kvm holds an enc_context_owner ref so its asid can't @@ -2994,7 +2996,7 @@ void sev_vm_destroy(struct kvm *kvm) * Free the mask even if the VM is not *currently* an SEV VM, as it may * have been an SEV VM prior to intra-host migration. */ - free_cpumask_var(sev->have_run_cpus); + sev_free_have_run_cpus(sev); if (!sev_guest(kvm)) return; @@ -3630,7 +3632,6 @@ int pre_sev_run(struct vcpu_svm *svm, int cpu) sd->sev_vmcbs[asid] = svm->vmcb; svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ASID; - vmcb_mark_dirty(svm->vmcb, VMCB_ASID); return 0; } diff --git a/arch/x86/kvm/svm/svm.c b/arch/x86/kvm/svm/svm.c index 7d59d301e1e5..fe0cba4731b2 100644 --- a/arch/x86/kvm/svm/svm.c +++ b/arch/x86/kvm/svm/svm.c @@ -673,16 +673,7 @@ static void clr_dr_intercepts(struct vcpu_svm *svm) static bool msr_write_intercepted(struct vcpu_svm *svm, u32 msr) { - /* - * For non-nested case: - * If the L01 MSR bitmap does not intercept the MSR, then we need to - * save it. - * - * For nested case: - * If the L02 MSR bitmap does not intercept the MSR, then we need to - * save it. - */ - void *msrpm = is_guest_mode(&svm->vcpu) ? svm->nested.msrpm : svm->msrpm; + void *msrpm = __va(__sme_clr(svm->vmcb->control.msrpm_base_pa)); return svm_test_msr_bitmap_write(msrpm, msr); } @@ -1902,8 +1893,7 @@ static void new_asid(struct vcpu_svm *svm, struct svm_cpu_data *sd) if (sd->next_asid > sd->max_asid) { ++sd->asid_generation; sd->next_asid = sd->min_asid; - svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ALL_ASID; - vmcb_mark_dirty(svm->vmcb, VMCB_ASID); + sd->flush_all_asids = true; } svm->current_vmcb->asid_generation = sd->asid_generation; @@ -4528,6 +4518,9 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags) svm->vmcb->control.asid = svm->asid; vmcb_mark_dirty(svm->vmcb, VMCB_ASID); } + if (this_cpu_ptr(&svm_data)->flush_all_asids) + svm->vmcb->control.tlb_ctl = TLB_CONTROL_FLUSH_ALL_ASID; + svm->vmcb->save.cr2 = vcpu->arch.cr2; if (guest_cpu_cap_has(vcpu, X86_FEATURE_ERAPS) && @@ -4618,16 +4611,23 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags) vcpu->arch.nested_run_pending = 0; } - svm->vmcb->control.tlb_ctl = TLB_CONTROL_DO_NOTHING; + if (!svm_is_vmrun_failure(svm->vmcb->control.exit_code)) { + this_cpu_ptr(&svm_data)->flush_all_asids = false; + svm->vmcb->control.tlb_ctl = TLB_CONTROL_DO_NOTHING; - /* - * Unconditionally mask off the CLEAR_RAP bit, the AND is just as cheap - * as the TEST+Jcc to avoid it. - */ - if (cpu_feature_enabled(X86_FEATURE_ERAPS)) - svm->vmcb->control.erap_ctl &= ~ERAP_CONTROL_CLEAR_RAP; + /* + * Unconditionally mask off the CLEAR_RAP bit, the AND is just + * as cheap as the TEST+Jcc to avoid it. + */ + if (cpu_feature_enabled(X86_FEATURE_ERAPS)) + svm->vmcb->control.erap_ctl &= ~ERAP_CONTROL_CLEAR_RAP; - vmcb_mark_all_clean(svm->vmcb); + vmcb_mark_all_clean(svm->vmcb); + + if (!msr_write_intercepted(svm, MSR_AMD64_PERF_CNTR_GLOBAL_CTL)) + rdmsrq(MSR_AMD64_PERF_CNTR_GLOBAL_CTL, + vcpu_to_pmu(vcpu)->global_ctrl); + } /* if exit due to PF check for async PF */ if (svm->vmcb->control.exit_code == SVM_EXIT_EXCP_BASE + PF_VECTOR) @@ -4636,9 +4636,6 @@ static __no_kcsan fastpath_t svm_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags) kvm_clear_available_registers(vcpu, SVM_REGS_LAZY_LOAD_SET); - if (!msr_write_intercepted(svm, MSR_AMD64_PERF_CNTR_GLOBAL_CTL)) - rdmsrq(MSR_AMD64_PERF_CNTR_GLOBAL_CTL, vcpu_to_pmu(vcpu)->global_ctrl); - trace_kvm_exit(vcpu, KVM_ISA_SVM); svm_complete_interrupts(vcpu); diff --git a/arch/x86/kvm/svm/svm.h b/arch/x86/kvm/svm/svm.h index e958943b8162..84f19026d3e8 100644 --- a/arch/x86/kvm/svm/svm.h +++ b/arch/x86/kvm/svm/svm.h @@ -376,6 +376,7 @@ struct svm_cpu_data { u32 next_asid; u32 min_asid; + bool flush_all_asids; bool bp_spec_reduce_set; struct vmcb *save_area; diff --git a/arch/x86/kvm/vmx/nested.c b/arch/x86/kvm/vmx/nested.c index 40c1a5f6fa8a..b25216862740 100644 --- a/arch/x86/kvm/vmx/nested.c +++ b/arch/x86/kvm/vmx/nested.c @@ -1754,6 +1754,9 @@ static void copy_vmcs12_to_shadow(struct vcpu_vmx *vmx) static void copy_enlightened_to_vmcs12(struct vcpu_vmx *vmx, u32 hv_clean_fields) { #ifdef CONFIG_KVM_HYPERV + const u64 runtime_controls = HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_GRP1 | + HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_GRP2 | + HV_VMX_ENLIGHTENED_CLEAN_FIELD_CONTROL_PROC; struct vmcs12 *vmcs12 = vmx->nested.cached_vmcs12; struct hv_enlightened_vmcs *evmcs = nested_vmx_evmcs(vmx); struct kvm_vcpu_hv *hv_vcpu = to_hv_vcpu(&vmx->vcpu); @@ -1762,6 +1765,9 @@ static void copy_enlightened_to_vmcs12(struct vcpu_vmx *vmx, u32 hv_clean_fields vmcs12->tpr_threshold = evmcs->tpr_threshold; vmcs12->guest_rip = evmcs->guest_rip; + if ((hv_clean_fields & runtime_controls) != runtime_controls) + vmx->nested.force_msr_bitmap_recalc = true; + if (unlikely(!(hv_clean_fields & HV_VMX_ENLIGHTENED_CLEAN_FIELD_ENLIGHTENMENTSCONTROL))) { hv_vcpu->nested.pa_page_gpa = evmcs->partition_assist_page; diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index 6c842e9191a5..00850a3a77d6 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -2728,11 +2728,6 @@ static tdx_vm_state_guard_t tdx_acquire_vm_state_locks(struct kvm *kvm) mutex_lock(&kvm->lock); - if (kvm->created_vcpus != atomic_read(&kvm->online_vcpus)) { - r = -EBUSY; - goto out_err; - } - r = kvm_lock_all_vcpus(kvm); if (r) goto out_err; diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index cf7fe835c4ad..c844def6dd95 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -791,7 +791,6 @@ struct kvm { /* The current active memslot set for each address space */ struct kvm_memslots __rcu *memslots[KVM_MAX_NR_ADDRESS_SPACES]; struct xarray vcpu_array; - DECLARE_BITMAP(vcpu_ids, KVM_MAX_VCPU_IDS); /* * Protected by slots_lock, but can be read outside if an * incorrect answer is acceptable. diff --git a/tools/testing/selftests/kvm/Makefile.kvm b/tools/testing/selftests/kvm/Makefile.kvm index 6a1482e3a286..f43549a56f0b 100644 --- a/tools/testing/selftests/kvm/Makefile.kvm +++ b/tools/testing/selftests/kvm/Makefile.kvm @@ -103,6 +103,7 @@ TEST_GEN_PROGS_x86 += x86/nested_tdp_fault_test TEST_GEN_PROGS_x86 += x86/nested_tsc_adjust_test TEST_GEN_PROGS_x86 += x86/nested_tsc_scaling_test TEST_GEN_PROGS_x86 += x86/nested_vmsave_vmload_test +TEST_GEN_PROGS_x86 += x86/nested_x2apic_test TEST_GEN_PROGS_x86 += x86/platform_info_test TEST_GEN_PROGS_x86 += x86/pmu_counters_test TEST_GEN_PROGS_x86 += x86/pmu_event_filter_test @@ -195,6 +196,7 @@ TEST_GEN_PROGS_arm64 += arm64/vgic_v5 TEST_GEN_PROGS_arm64 += arm64/vpmu_counter_access TEST_GEN_PROGS_arm64 += arm64/no-vgic TEST_GEN_PROGS_arm64 += arm64/idreg-idst +TEST_GEN_PROGS_arm64 += arm64/hidden_features TEST_GEN_PROGS_arm64 += arm64/kvm-uuid TEST_GEN_PROGS_arm64 += access_tracking_perf_test TEST_GEN_PROGS_arm64 += arch_timer diff --git a/tools/testing/selftests/kvm/arm64/hidden_features.c b/tools/testing/selftests/kvm/arm64/hidden_features.c new file mode 100644 index 000000000000..194d746e7605 --- /dev/null +++ b/tools/testing/selftests/kvm/arm64/hidden_features.c @@ -0,0 +1,184 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * hidden_features - Check that a feature's instruction runs in the guest when + * its ID register field is advertised, and is UNDEFINED when userspace clears + * the field. + * + * Copyright (c) 2026 Google LLC + * Author: Fuad Tabba <fuad.tabba@linux.dev> + */ +#include "kvm_util.h" +#include "processor.h" +#include "test_util.h" + +static volatile bool undef; + +static void guest_tlbi_os(void) +{ + /* tlbi vmalle1os */ + asm volatile("sys #0, c8, c1, #0\n\tdsb ish\n\tisb" ::: "memory"); +} + +static void guest_mops(void) +{ + register u64 *d asm("x0"); + register u64 n asm("x1"); + register u64 s asm("x2"); + u64 buf[8]; + + d = buf; + n = sizeof(buf); + s = 0; + /* setp [x0]!, x1!, x2; setm; sete */ + asm volatile(".inst 0x19c20420\n\t.inst 0x19c24420\n\t.inst 0x19c28420" + : "+r"(d), "+r"(n) : "r"(s) : "cc", "memory"); +} + +static void guest_tcr2(void) +{ + read_sysreg_s(SYS_TCR2_EL1); +} + +static void guest_fpmr(void) +{ + read_sysreg_s(SYS_FPMR); +} + +struct feature { + const char *name; + u64 id_reg; + u64 mask; + u8 shift; + u64 min; + void (*insn)(void); + bool (*trappable)(struct kvm_vcpu *vcpu); +}; + +/* Without FGT, KVM traps a hidden TLBI OS only through HCR_EL2.TTLBOS (FEAT_EVT2). */ +static bool tlbi_os_trappable(struct kvm_vcpu *vcpu) +{ + u64 mmfr0 = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(SYS_ID_AA64MMFR0_EL1)); + u64 mmfr2 = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(SYS_ID_AA64MMFR2_EL1)); + + return SYS_FIELD_GET(ID_AA64MMFR0_EL1, FGT, mmfr0) >= ID_AA64MMFR0_EL1_FGT_IMP || + SYS_FIELD_GET(ID_AA64MMFR2_EL1, EVT, mmfr2) >= ID_AA64MMFR2_EL1_EVT_TTLBxS; +} + +#define FEATURE(n, reg, field, min_val, fn, trap) \ +{ \ + .name = n, \ + .id_reg = SYS_##reg, \ + .mask = reg##_##field##_MASK, \ + .shift = reg##_##field##_SHIFT, \ + .min = reg##_##field##_##min_val, \ + .insn = fn, \ + .trappable = trap, \ +} + +static const struct feature features[] = { + FEATURE("TLBI OS", ID_AA64ISAR0_EL1, TLB, OS, guest_tlbi_os, tlbi_os_trappable), + FEATURE("MOPS", ID_AA64ISAR2_EL1, MOPS, IMP, guest_mops, NULL), + FEATURE("TCR2_EL1", ID_AA64MMFR3_EL1, TCRX, IMP, guest_tcr2, NULL), + FEATURE("FPMR", ID_AA64PFR2_EL1, FPMR, IMP, guest_fpmr, NULL), +}; + +static void guest_code(const struct feature *feat) +{ + undef = false; + feat->insn(); + GUEST_SYNC(undef); + GUEST_DONE(); +} + +static void guest_undef_handler(struct ex_regs *regs) +{ + undef = true; + regs->pc += 4; +} + +static bool run(const struct feature *feat, bool hide) +{ + struct kvm_vcpu *vcpu; + struct kvm_vm *vm; + struct ucall uc; + bool got = false; + u64 val; + + vm = vm_create_with_one_vcpu(&vcpu, (void *)guest_code); + vm_init_descriptor_tables(vm); + vcpu_init_descriptor_tables(vcpu); + vm_install_sync_handler(vm, VECTOR_SYNC_CURRENT, ESR_ELx_EC_UNKNOWN, guest_undef_handler); + vcpu_args_set(vcpu, 1, feat); + + if (hide) { + val = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg)); + vcpu_set_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg), val & ~feat->mask); + } + + for (;;) { + vcpu_run(vcpu); + switch (get_ucall(vcpu, &uc)) { + case UCALL_SYNC: + got = uc.args[1]; + break; + case UCALL_ABORT: + REPORT_GUEST_ASSERT(uc); + break; + case UCALL_DONE: + kvm_vm_free(vm); + return got; + default: + TEST_FAIL("Unknown ucall %lu", uc.cmd); + } + } +} + +static void probe_feature(const struct feature *feat, bool *present, bool *trappable) +{ + struct kvm_vcpu *vcpu; + struct kvm_vm *vm; + u64 val; + + vm = vm_create_with_one_vcpu(&vcpu, NULL); + val = vcpu_get_reg(vcpu, KVM_ARM64_SYS_REG(feat->id_reg)); + *present = ((val & feat->mask) >> feat->shift) >= feat->min; + *trappable = !feat->trappable || feat->trappable(vcpu); + kvm_vm_free(vm); +} + +int main(void) +{ + const struct feature *feat; + bool present, trappable; + int i; + + test_disable_default_vgic(); + + ksft_print_header(); + ksft_set_plan(ARRAY_SIZE(features) * 2); + + for (i = 0; i < ARRAY_SIZE(features); i++) { + feat = &features[i]; + + probe_feature(feat, &present, &trappable); + if (!present) { + ksft_test_result_skip("%s advertised, not supported\n", feat->name); + ksft_test_result_skip("%s hidden, not supported\n", feat->name); + continue; + } + + if (run(feat, false)) + ksft_test_result_fail("%s advertised, UNDEF\n", feat->name); + else + ksft_test_result_pass("%s advertised\n", feat->name); + + if (!trappable) + ksft_test_result_skip("%s hidden, not trappable\n", feat->name); + else if (run(feat, true)) + ksft_test_result_pass("%s hidden\n", feat->name); + else + ksft_test_result_fail("%s hidden, no UNDEF\n", feat->name); + } + + ksft_finished(); +} diff --git a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c new file mode 100644 index 000000000000..ce204ce29a9c --- /dev/null +++ b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c @@ -0,0 +1,222 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include "test_util.h" +#include "kvm_util.h" +#include "processor.h" +#include "vmx.h" +#include "svm_util.h" + +/* + * Use the kernel's posted interrupt vectors to minimize the risk of crashing + * the host if KVM is buggy. Note, the vectors aren't set in stone, ideally + * these will be kept up-to-date if the kernel vectors change, but it's "fine" + * if they are stale. + */ +#define POSTED_INTR_VECTOR 0xf2 +#define POSTED_INTR_WAKEUP_VECTOR 0xf1 +#define POSTED_INTR_NESTED_VECTOR 0xf0 + +static bool inhibit_apicv; +static volatile unsigned int nr_irqs; + +static void guest_irq_handler(struct ex_regs *regs) +{ + nr_irqs++; + x2apic_write_reg(APIC_EOI, 0); +} + +static void l2_guest_code(void) +{ + if (inhibit_apicv) + wrmsr(MSR_IA32_APICBASE, rdmsr(MSR_IA32_APICBASE) & GENMASK_ULL(11, 0)); + + for (;;) { + x2apic_write_reg(APIC_TASKPRI, 0xf0); + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0); + + asm volatile("cpuid" ::: "eax", "ebx", "ecx", "edx"); + } +} + +static void l1_svm_code(struct svm_test_data *svm) +{ + struct vmcb_control_area *ctrl = &svm->vmcb->control; + + generic_svm_setup(svm, l2_guest_code); + ctrl->intercept |= BIT_ULL(INTERCEPT_CPUID) | BIT_ULL(INTERCEPT_MSR_PROT); + + run_guest(svm->vmcb, svm->vmcb_gpa); + GUEST_ASSERT_EQ(ctrl->exit_code, SVM_EXIT_CPUID); + + stgi(); + x2apic_write_reg(APIC_TASKPRI, 0); +} + +static void l1_vmx_code(struct vmx_pages *vmx, struct hyperv_test_pages *hv_pages) +{ + u64 control; + + if (hv_pages) { + wrmsr(HV_X64_MSR_GUEST_OS_ID, HYPERV_LINUX_OS_ID); + enable_vp_assist(hv_pages->vp_assist_gpa, hv_pages->vp_assist); + evmcs_enable(); + } + + GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true); + + if (hv_pages) { + GUEST_ASSERT(load_evmcs(hv_pages)); + current_evmcs->hv_enlightenments_control.msr_bitmap = 1; + } else { + GUEST_ASSERT(load_vmcs(vmx)); + } + + prepare_vmcs(vmx, NULL); + GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, (unsigned long)l2_guest_code), 0); + + control = vmreadz(PIN_BASED_VM_EXEC_CONTROL); + control |= PIN_BASED_EXT_INTR_MASK; + vmwrite(PIN_BASED_VM_EXEC_CONTROL, control); + + control = vmreadz(CPU_BASED_VM_EXEC_CONTROL); + control |= CPU_BASED_USE_MSR_BITMAPS | CPU_BASED_TPR_SHADOW; + GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0); + + control = vmreadz(SECONDARY_VM_EXEC_CONTROL); + control |= SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE | + SECONDARY_EXEC_APIC_REGISTER_VIRT | + SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY; + control &= (rdmsr(MSR_IA32_VMX_PROCBASED_CTLS2) >> 32); + GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0); + + GUEST_ASSERT(!vmlaunch()); + GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID); + GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, + vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0); +} + +static void l1_vmx_code_part2(void) +{ + u64 control; + + control = vmreadz(CPU_BASED_VM_EXEC_CONTROL); + control &= ~CPU_BASED_TPR_SHADOW; + GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0); + + control = vmreadz(SECONDARY_VM_EXEC_CONTROL); + control &= ~(SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE | + SECONDARY_EXEC_APIC_REGISTER_VIRT | + SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY); + GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0); + + GUEST_ASSERT(!vmresume()); + GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID); + GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, + vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0); +} + +static void l1_test_x2apic_intercepts(void) +{ + GUEST_ASSERT_EQ(nr_irqs, 0); + + sti_nop(); + + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0); + + x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_VECTOR); + GUEST_ASSERT_EQ(nr_irqs, 0); + + x2apic_write_reg(APIC_TASKPRI, 0xff); + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xff); + x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_WAKEUP_VECTOR); + GUEST_ASSERT_EQ(nr_irqs, 0); + + x2apic_write_reg(APIC_TASKPRI, 0); + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0); + + x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_NESTED_VECTOR); + GUEST_ASSERT_EQ(nr_irqs, 3); + + nr_irqs = 0; +} + +static void l1_guest_code(void *test_data, void *hv_pages) +{ + x2apic_enable(); + + if (this_cpu_has(X86_FEATURE_SVM)) + l1_svm_code(test_data); + else + l1_vmx_code(test_data, hv_pages); + + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0); + x2apic_write_reg(APIC_TASKPRI, 0xf0); + + l1_test_x2apic_intercepts(); + + if (this_cpu_has(X86_FEATURE_VMX)) { + l1_vmx_code_part2(); + l1_test_x2apic_intercepts(); + } + + GUEST_DONE(); +} + +static void test_x2apic_intercepts(bool with_inhibit_apicv, bool use_evmcs) +{ + gva_t nested_test_data_gva, hv_pages_gva = 0; + struct kvm_vcpu *vcpu; + struct kvm_vm *vm; + struct ucall uc; + + inhibit_apicv = with_inhibit_apicv; + + vm = vm_create_with_one_vcpu(&vcpu, l1_guest_code); + vm_install_exception_handler(vm, POSTED_INTR_VECTOR, guest_irq_handler); + vm_install_exception_handler(vm, POSTED_INTR_WAKEUP_VECTOR, guest_irq_handler); + vm_install_exception_handler(vm, POSTED_INTR_NESTED_VECTOR, guest_irq_handler); + + sync_global_to_guest(vm, inhibit_apicv); + + if (kvm_cpu_has(X86_FEATURE_SVM)) + vcpu_alloc_svm(vm, &nested_test_data_gva); + else + vcpu_alloc_vmx(vm, &nested_test_data_gva); + + if (use_evmcs) { + vcpu_set_hv_cpuid(vcpu); + vcpu_enable_evmcs(vcpu); + + vcpu_alloc_hyperv_test_pages(vm, &hv_pages_gva); + } + + vcpu_args_set(vcpu, 2, nested_test_data_gva, hv_pages_gva); + + vcpu_run(vcpu); + + TEST_ASSERT_KVM_EXIT_REASON(vcpu, KVM_EXIT_IO); + + switch (get_ucall(vcpu, &uc)) { + case UCALL_DONE: + break; + case UCALL_ABORT: + REPORT_GUEST_ASSERT(uc); + break; + default: + TEST_FAIL("Expected DONE, got unexpected ucall %lu", uc.cmd); + } + + kvm_vm_free(vm); +} + +int main(int argc, char *argv[]) +{ + TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_SVM) || kvm_cpu_has(X86_FEATURE_VMX)); + + test_x2apic_intercepts(true, false); + test_x2apic_intercepts(false, false); + + if (kvm_has_cap(KVM_CAP_HYPERV_ENLIGHTENED_VMCS)) { + test_x2apic_intercepts(true, true); + test_x2apic_intercepts(false, true); + } +} diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 85f42289748d..5a782849521c 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -1363,6 +1363,9 @@ int kvm_trylock_all_vcpus(struct kvm *kvm) lockdep_assert_held(&kvm->lock); + if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm))) + return -EBUSY; + kvm_for_each_vcpu(i, vcpu, kvm) if (!mutex_trylock_nest_lock(&vcpu->mutex, &kvm->lock)) goto out_unlock; @@ -1386,6 +1389,9 @@ int kvm_lock_all_vcpus(struct kvm *kvm) lockdep_assert_held(&kvm->lock); + if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm))) + return -EBUSY; + kvm_for_each_vcpu(i, vcpu, kvm) { r = mutex_lock_killable_nest_lock(&vcpu->mutex, &kvm->lock); if (r) @@ -4182,6 +4188,8 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id) struct kvm_vcpu *vcpu; struct page *page; + guard(mutex)(&kvm->lock); + /* * KVM tracks vCPU IDs as 'int', be kind to userspace and reject * too-large values instead of silently truncating. @@ -4194,26 +4202,17 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id) if (id >= KVM_MAX_VCPU_IDS) return -EINVAL; - mutex_lock(&kvm->lock); - if (kvm->created_vcpus >= kvm->max_vcpus) { - mutex_unlock(&kvm->lock); + if (kvm->created_vcpus >= kvm->max_vcpus) return -EINVAL; - } - if (test_bit(id, kvm->vcpu_ids)) { - mutex_unlock(&kvm->lock); + if (kvm_get_vcpu_by_id(kvm, id)) return -EEXIST; - } r = kvm_arch_vcpu_precreate(kvm, id); - if (r) { - mutex_unlock(&kvm->lock); + if (r) return r; - } kvm->created_vcpus++; - __set_bit(id, kvm->vcpu_ids); - mutex_unlock(&kvm->lock); vcpu = kmem_cache_zalloc(kvm_vcpu_cache, GFP_KERNEL_ACCOUNT); if (!vcpu) { @@ -4244,13 +4243,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id) goto arch_vcpu_destroy; } - mutex_lock(&kvm->lock); - - if (WARN_ON_ONCE(kvm_get_vcpu_by_id(kvm, id))) { - r = -EEXIST; - goto unlock_vcpu_destroy; - } - /* * Set the vCPU's index *before* the vCPU is reachable by other tasks. * Unwind the index back to -1 on failure so that KVM can use the index @@ -4284,7 +4276,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id) atomic_inc(&kvm->online_vcpus); mutex_unlock(&vcpu->mutex); - mutex_unlock(&kvm->lock); kvm_arch_vcpu_postcreate(vcpu); kvm_create_vcpu_debugfs(vcpu); return r; @@ -4295,7 +4286,6 @@ kvm_put_xa_erase: xa_erase(&kvm->vcpu_array, vcpu->vcpu_idx); unlock_vcpu_destroy: vcpu->vcpu_idx = -1; - mutex_unlock(&kvm->lock); kvm_dirty_ring_free(&vcpu->dirty_ring); arch_vcpu_destroy: kvm_arch_vcpu_destroy(vcpu); @@ -4304,10 +4294,7 @@ vcpu_free_run_page: vcpu_free: kmem_cache_free(kvm_vcpu_cache, vcpu); vcpu_decrement: - mutex_lock(&kvm->lock); kvm->created_vcpus--; - __clear_bit(id, kvm->vcpu_ids); - mutex_unlock(&kvm->lock); return r; } |
