summaryrefslogtreecommitdiff
path: root/virt/kvm
diff options
context:
space:
mode:
authorSean Christopherson <seanjc@google.com>2026-09-29 15:25:07 -0700
committerSean Christopherson <seanjc@google.com>2026-09-29 15:25:07 -0700
commit703dd419812dc0bd6e284ed90d57fdda3a642f37 (patch)
treed6159b617881d70e52cfc3b88947e0dab63cdc7f /virt/kvm
parentc45a832994cbeed83d695ac489f94996b3fd482a (diff)
parent56329611a670eedc78e1870ae96c2b7ccc580709 (diff)
downloadlinux-next-703dd419812dc0bd6e284ed90d57fdda3a642f37.tar.gz
linux-next-703dd419812dc0bd6e284ed90d57fdda3a642f37.zip
Merge branch 'fixes'
* fixes: (67 commits) KVM: x86: Use active memslots for the per-vCPU MMIO cache KVM: SEV: Fix page dirtying in sev_gmem_post_populate() KVM: SVM: Clear AVIC Physical ID table entry if vCPU creation fails KVM: SVM: Use "is AVIC-addressable" helper to sanity check load()/put() KVM: SVM: Add paranoid helper for checking if vCPU is AVIC-addressable KVM: selftests: Extend nested x2APIC test to validate using eVMCS for vmcs12 KVM: selftests: Extend nested x2APIC test to validate disabling x2APIC virt KVM: selftests: Verify that L0's TPR doesn't get clobbered KVM: selftests: Run the nested x2APIC with and without APICv being inhibited in L2 KVM: selftests: Add x2APIC MSR test for inhibiting APICv while nested KVM: nVMX: Force MSR bitmap refresh if runtime eVMCS controls are modified KVM: SVM: Use the active VMCB's MSR bitmap when checking if MSR is intercepted KVM: SVM: Sync guest's PERF_CNTR_GLOBAL_CTL from h/w only on successful VMRUN KVM: SVM: Don't mark ASID fields as dirty when setting control.tlb_ctl KVM: SVM: Update control fields on #VMEXIT if and only if VMRUN succeeded KVM: SVM: Preserve TLB control (i.e. pending TLB flush) on failed VMRUN KVM: x86/mmu: Bail from shadow walks if the root is invalid or a dummy KVM: WARN if vCPU creation is in-progress when locking all vCPUs Revert "KVM: Check for duplicate vcpu_id as early as possible" KVM: Move check for existing vCPU ID to the top of vCPU creation ...
Diffstat (limited to 'virt/kvm')
-rw-r--r--virt/kvm/kvm_main.c70
1 files changed, 40 insertions, 30 deletions
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 1dd3259eed6c..dcc0f2cdc47e 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -1135,7 +1135,7 @@ static struct kvm *kvm_create_vm(unsigned long type, const char *fdname)
rcuwait_init(&kvm->mn_memslots_update_rcuwait);
xa_init(&kvm->vcpu_array);
#ifdef CONFIG_KVM_VM_MEMORY_ATTRIBUTES
- xa_init(&kvm->mem_attr_array);
+ xa_init_flags(&kvm->mem_attr_array, XA_FLAGS_ACCOUNT);
#endif
INIT_LIST_HEAD(&kvm->gpc_list);
@@ -1381,6 +1381,9 @@ int kvm_trylock_all_vcpus(struct kvm *kvm)
lockdep_assert_held(&kvm->lock);
+ if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm)))
+ return -EBUSY;
+
kvm_for_each_vcpu(i, vcpu, kvm)
if (!mutex_trylock_nest_lock(&vcpu->mutex, &kvm->lock))
goto out_unlock;
@@ -1404,6 +1407,9 @@ int kvm_lock_all_vcpus(struct kvm *kvm)
lockdep_assert_held(&kvm->lock);
+ if (WARN_ON_ONCE(kvm_is_vcpu_creation_in_progress(kvm)))
+ return -EBUSY;
+
kvm_for_each_vcpu(i, vcpu, kvm) {
r = mutex_lock_killable_nest_lock(&vcpu->mutex, &kvm->lock);
if (r)
@@ -2493,14 +2499,36 @@ bool kvm_range_has_vm_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
return (kvm_get_vm_memory_attributes(kvm, start) & mask) == attrs;
guard(rcu)();
- if (!attrs)
- return !xas_find(&xas, end - 1);
+ /*
+ * Lookup the entry for each index instead of iterating over the xarray
+ * as KVM deletes/nullifies entries to represent "no attributes", and
+ * the xas index is effectively invalid when no entry is found. I.e.
+ * matching non-zero attributes for *every* entry effectively requires
+ * a manually lookup for each index.
+ *
+ * Skip pre-allocated, reserved entries, or restart the lookup if the
+ * xarray was concurrently modified, via xas_retry() ("retry" means the
+ * entry holds an internal xarray value, i.e. is either invalid or NULL
+ * from the caller's perspective).
+ *
+ * Use xas_next() when looking for non-zero attributes to optimize for
+ * the case where the start of the range (or the entire range) doesn't
+ * have any attributes, as xas_next() returns literally the next entry,
+ * whereas xas_next_entry() returns the next non-NULL entry (bounded by
+ * a maximum index).
+ */
for (index = start; index < end; index++) {
do {
- entry = xas_next(&xas);
+ entry = attrs ? xas_next(&xas) :
+ xas_next_entry(&xas, end - 1);
} while (xas_retry(&xas, entry));
+ if (!entry)
+ return !attrs;
+
+ WARN_ON_ONCE(!xa_to_value(entry));
+
if (xas.xa_index != index ||
(xa_to_value(entry) & mask) != attrs)
return false;
@@ -2617,9 +2645,10 @@ static int kvm_set_vm_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
/*
* Reserve memory ahead of time to avoid having to deal with failures
- * partway through setting the new attributes.
+ * partway through setting the new attributes. Storing NULL never
+ * allocates, so no reservations are needed when clearing.
*/
- for (i = start; i < end; i++) {
+ for (i = start; entry && i < end; i++) {
r = xa_reserve(&kvm->mem_attr_array, i, GFP_KERNEL_ACCOUNT);
if (r)
goto out_unlock;
@@ -4222,6 +4251,8 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
struct kvm_vcpu *vcpu;
struct page *page;
+ guard(mutex)(&kvm->lock);
+
/*
* KVM tracks vCPU IDs as 'int', be kind to userspace and reject
* too-large values instead of silently truncating.
@@ -4234,26 +4265,17 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
if (id >= KVM_MAX_VCPU_IDS)
return -EINVAL;
- mutex_lock(&kvm->lock);
- if (kvm->created_vcpus >= kvm->max_vcpus) {
- mutex_unlock(&kvm->lock);
+ if (kvm->created_vcpus >= kvm->max_vcpus)
return -EINVAL;
- }
- if (test_bit(id, kvm->vcpu_ids)) {
- mutex_unlock(&kvm->lock);
+ if (kvm_get_vcpu_by_id(kvm, id))
return -EEXIST;
- }
r = kvm_arch_vcpu_precreate(kvm, id);
- if (r) {
- mutex_unlock(&kvm->lock);
+ if (r)
return r;
- }
kvm->created_vcpus++;
- __set_bit(id, kvm->vcpu_ids);
- mutex_unlock(&kvm->lock);
vcpu = kmem_cache_zalloc(kvm_vcpu_cache, GFP_KERNEL_ACCOUNT);
if (!vcpu) {
@@ -4284,13 +4306,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
goto arch_vcpu_destroy;
}
- mutex_lock(&kvm->lock);
-
- if (WARN_ON_ONCE(kvm_get_vcpu_by_id(kvm, id))) {
- r = -EEXIST;
- goto unlock_vcpu_destroy;
- }
-
/*
* Set the vCPU's index *before* the vCPU is reachable by other tasks.
* Unwind the index back to -1 on failure so that KVM can use the index
@@ -4324,7 +4339,6 @@ static int kvm_vm_ioctl_create_vcpu(struct kvm *kvm, unsigned long id)
atomic_inc(&kvm->online_vcpus);
mutex_unlock(&vcpu->mutex);
- mutex_unlock(&kvm->lock);
kvm_arch_vcpu_postcreate(vcpu);
kvm_create_vcpu_debugfs(vcpu);
return r;
@@ -4335,7 +4349,6 @@ kvm_put_xa_erase:
xa_erase(&kvm->vcpu_array, vcpu->vcpu_idx);
unlock_vcpu_destroy:
vcpu->vcpu_idx = -1;
- mutex_unlock(&kvm->lock);
kvm_dirty_ring_free(&vcpu->dirty_ring);
arch_vcpu_destroy:
kvm_arch_vcpu_destroy(vcpu);
@@ -4344,10 +4357,7 @@ vcpu_free_run_page:
vcpu_free:
kmem_cache_free(kvm_vcpu_cache, vcpu);
vcpu_decrement:
- mutex_lock(&kvm->lock);
kvm->created_vcpus--;
- __clear_bit(id, kvm->vcpu_ids);
- mutex_unlock(&kvm->lock);
return r;
}