summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorPaolo Bonzini <pbonzini@redhat.com>2026-09-26 00:40:07 -0400
committerPaolo Bonzini <pbonzini@redhat.com>2026-09-26 00:40:07 -0400
commitc2f24f140c2ee6c00775c2a93c6ac931acec2b60 (patch)
treefd13802556683cc72528d35173ccbc4429742f73
parent93de2a6a4b91b72607136dd656edf03fb399d27f (diff)
parent119e9db233ad0f4cbd4cfb9f9385a7bca378b37d (diff)
downloadlinux-c2f24f140c2ee6c00775c2a93c6ac931acec2b60.tar.gz
linux-c2f24f140c2ee6c00775c2a93c6ac931acec2b60.zip
Merge tag 'kvm-x86-fixes-7.3-rc5' of https://github.com/kvm-x86/linux into HEAD
KVM fixes for 7.3-rcN - Fix a brown paper bag bug where KVM would incorrectly treat Intel PMU MSRs as valid on AMD. - Fix a regression in the hardware disable selftest where it checked the wrong macro when detecting glibc support (breaks at least musl). - Never clear KVM_REQ_VM_DEAD so that dead VMs stay dead, which is especially important for KVM_BUG_ON() flows, which often guard more dangerous bugs. - Re-pend GET_NESTED_STATE_PAGES if getting the pages fails, to fix a bug where KVM would let userspace run a broken setup with stale vmcs12 pages. - Fix a class of bugs where KVM would fail to fill kvm_run exit fields if getting nested pages failed. - Treat reserved entries in the memory attributes xarray as "no attributes", to fix false positives when checking for mixed attributes. - Fix memcg accounting for the memory attributes xarray (the xarray library subtly requires the xarray to be configured for accounting upfront; the gfp flags taken at runtime are used only rarely). - Don't pre-reserve xarray entries when storing empty attributes, as storing NULL must not require memory allocation (KVM and other subsystems heavily rely on this behavior).
-rw-r--r--arch/arm64/kvm/arm.c2
-rw-r--r--arch/x86/kvm/mmu/mmu.c2
-rw-r--r--arch/x86/kvm/pmu.c8
-rw-r--r--arch/x86/kvm/svm/nested.c7
-rw-r--r--arch/x86/kvm/vmx/nested.c15
-rw-r--r--arch/x86/kvm/vmx/pmu_intel.c3
-rw-r--r--arch/x86/kvm/vmx/tdx.c2
-rw-r--r--arch/x86/kvm/x86.c6
-rw-r--r--include/linux/kvm_host.h9
-rw-r--r--tools/testing/selftests/kvm/hardware_disable_test.c6
-rw-r--r--virt/kvm/kvm_main.c35
11 files changed, 56 insertions, 39 deletions
diff --git a/arch/arm64/kvm/arm.c b/arch/arm64/kvm/arm.c
index eaf583b77193..0576c2022ef5 100644
--- a/arch/arm64/kvm/arm.c
+++ b/arch/arm64/kvm/arm.c
@@ -1133,7 +1133,7 @@ static int kvm_vcpu_suspend(struct kvm_vcpu *vcpu)
static int check_vcpu_requests(struct kvm_vcpu *vcpu)
{
if (kvm_request_pending(vcpu)) {
- if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu))
+ if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu))
return -EIO;
if (kvm_check_request(KVM_REQ_SLEEP, vcpu))
diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
index 064ecc33b926..8e62476e477b 100644
--- a/arch/x86/kvm/mmu/mmu.c
+++ b/arch/x86/kvm/mmu/mmu.c
@@ -5058,7 +5058,7 @@ static int kvm_tdp_page_prefault(struct kvm_vcpu *vcpu, gpa_t gpa,
if (signal_pending(current))
return -EINTR;
- if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu))
+ if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu))
return -EIO;
cond_resched();
diff --git a/arch/x86/kvm/pmu.c b/arch/x86/kvm/pmu.c
index a7d60c8785cd..d2fd47ee5ec8 100644
--- a/arch/x86/kvm/pmu.c
+++ b/arch/x86/kvm/pmu.c
@@ -823,14 +823,6 @@ void kvm_pmu_deliver_pmi(struct kvm_vcpu *vcpu)
bool kvm_pmu_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr)
{
- switch (msr) {
- case MSR_CORE_PERF_GLOBAL_STATUS:
- case MSR_CORE_PERF_GLOBAL_CTRL:
- case MSR_CORE_PERF_GLOBAL_OVF_CTRL:
- return kvm_pmu_has_perf_global_ctrl(vcpu_to_pmu(vcpu));
- default:
- break;
- }
return kvm_pmu_call(msr_idx_to_pmc)(vcpu, msr) ||
kvm_pmu_call(is_valid_msr)(vcpu, msr);
}
diff --git a/arch/x86/kvm/svm/nested.c b/arch/x86/kvm/svm/nested.c
index 73f37b050d0a..f9090b601efa 100644
--- a/arch/x86/kvm/svm/nested.c
+++ b/arch/x86/kvm/svm/nested.c
@@ -2125,13 +2125,8 @@ static bool svm_get_nested_state_pages(struct kvm_vcpu *vcpu)
return false;
}
- if (!nested_svm_merge_msrpm(vcpu)) {
- vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
- vcpu->run->internal.suberror =
- KVM_INTERNAL_ERROR_EMULATION;
- vcpu->run->internal.ndata = 0;
+ if (!nested_svm_merge_msrpm(vcpu))
return false;
- }
if (kvm_hv_verify_vp_assist(vcpu))
return false;
diff --git a/arch/x86/kvm/vmx/nested.c b/arch/x86/kvm/vmx/nested.c
index 151873407abd..40c1a5f6fa8a 100644
--- a/arch/x86/kvm/vmx/nested.c
+++ b/arch/x86/kvm/vmx/nested.c
@@ -3465,10 +3465,6 @@ static bool nested_get_vmcs12_pages(struct kvm_vcpu *vcpu)
} else {
pr_debug_ratelimited("%s: no backing for APIC-access address in vmcs12\n",
__func__);
- vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
- vcpu->run->internal.suberror =
- KVM_INTERNAL_ERROR_EMULATION;
- vcpu->run->internal.ndata = 0;
return false;
}
}
@@ -3539,11 +3535,6 @@ static bool vmx_get_nested_state_pages(struct kvm_vcpu *vcpu)
if (!nested_get_evmcs_page(vcpu)) {
pr_debug_ratelimited("%s: enlightened vmptrld failed\n",
__func__);
- vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
- vcpu->run->internal.suberror =
- KVM_INTERNAL_ERROR_EMULATION;
- vcpu->run->internal.ndata = 0;
-
return false;
}
#endif
@@ -3915,8 +3906,12 @@ static int nested_vmx_run(struct kvm_vcpu *vcpu, bool launch)
vmentry_failed:
vcpu->arch.nested_run_pending = 0;
- if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR)
+ if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR) {
+ vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
+ vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION;
+ vcpu->run->internal.ndata = 0;
return 0;
+ }
if (status == NVMX_VMENTRY_VMEXIT)
return 1;
WARN_ON_ONCE(status != NVMX_VMENTRY_VMFAIL);
diff --git a/arch/x86/kvm/vmx/pmu_intel.c b/arch/x86/kvm/vmx/pmu_intel.c
index bfa8612fb450..70a8c4816135 100644
--- a/arch/x86/kvm/vmx/pmu_intel.c
+++ b/arch/x86/kvm/vmx/pmu_intel.c
@@ -187,6 +187,9 @@ static bool intel_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr)
int ret;
switch (msr) {
+ case MSR_CORE_PERF_GLOBAL_STATUS:
+ case MSR_CORE_PERF_GLOBAL_CTRL:
+ case MSR_CORE_PERF_GLOBAL_OVF_CTRL:
case MSR_CORE_PERF_FIXED_CTR_CTRL:
return kvm_pmu_has_perf_global_ctrl(pmu);
case MSR_IA32_PEBS_ENABLE:
diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c
index b272c20586a7..6c842e9191a5 100644
--- a/arch/x86/kvm/vmx/tdx.c
+++ b/arch/x86/kvm/vmx/tdx.c
@@ -1995,7 +1995,7 @@ static int tdx_handle_ept_violation(struct kvm_vcpu *vcpu)
if (kvm_vcpu_has_events(vcpu) || signal_pending(current))
break;
- if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) {
+ if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) {
ret = -EIO;
break;
}
diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c
index 79468ddfe473..aad065d035fb 100644
--- a/arch/x86/kvm/x86.c
+++ b/arch/x86/kvm/x86.c
@@ -8060,7 +8060,7 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu)
bool req_immediate_exit = false;
if (kvm_request_pending(vcpu)) {
- if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) {
+ if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) {
r = -EIO;
goto out;
}
@@ -8072,6 +8072,10 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu)
if (kvm_check_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu)) {
if (unlikely(!kvm_nested_call(get_nested_state_pages)(vcpu))) {
+ vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
+ vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION;
+ vcpu->run->internal.ndata = 0;
+ kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu);
r = 0;
goto out;
}
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index 03bfc92864b6..cf7fe835c4ad 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -2324,13 +2324,18 @@ static inline bool kvm_test_request(int req, struct kvm_vcpu *vcpu)
return test_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
-static inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
+static __always_inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
{
+ BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
+
clear_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
-static inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
+static __always_inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
{
+ /* Once a VM is dead, it needs to stay dead. */
+ BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
+
if (kvm_test_request(req, vcpu)) {
kvm_clear_request(req, vcpu);
diff --git a/tools/testing/selftests/kvm/hardware_disable_test.c b/tools/testing/selftests/kvm/hardware_disable_test.c
index 43a36ef3ead8..1c20892d6782 100644
--- a/tools/testing/selftests/kvm/hardware_disable_test.c
+++ b/tools/testing/selftests/kvm/hardware_disable_test.c
@@ -37,7 +37,7 @@ static void *run_vcpu(void *arg)
struct kvm_vcpu *vcpu = arg;
struct kvm_run *run = vcpu->run;
-#ifndef _GNU_SOURCE
+#ifndef __GLIBC__
kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
#endif
@@ -51,7 +51,7 @@ static void *sleeping_thread(void *arg)
{
int fd;
-#ifndef _GNU_SOURCE
+#ifndef __GLIBC__
kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
#endif
@@ -71,7 +71,7 @@ static void run_test(u32 run)
u32 i, j;
TEST_ASSERT_EQ(pthread_attr_init(&attr), 0);
-#ifdef _GNU_SOURCE
+#ifdef __GLIBC__
TEST_ASSERT_EQ(pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &threads_cpu_set), 0);
#endif
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 65eb26a0520d..85f42289748d 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -1117,7 +1117,7 @@ static struct kvm *kvm_create_vm(unsigned long type, const char *fdname)
rcuwait_init(&kvm->mn_memslots_update_rcuwait);
xa_init(&kvm->vcpu_array);
#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES
- xa_init(&kvm->mem_attr_array);
+ xa_init_flags(&kvm->mem_attr_array, XA_FLAGS_ACCOUNT);
#endif
INIT_LIST_HEAD(&kvm->gpc_list);
@@ -2447,14 +2447,36 @@ bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
return (kvm_get_memory_attributes(kvm, start) & mask) == attrs;
guard(rcu)();
- if (!attrs)
- return !xas_find(&xas, end - 1);
+ /*
+ * Lookup the entry for each index instead of iterating over the xarray
+ * as KVM deletes/nullifies entries to represent "no attributes", and
+ * the xas index is effectively invalid when no entry is found. I.e.
+ * matching non-zero attributes for *every* entry effectively requires
+ * a manually lookup for each index.
+ *
+ * Skip pre-allocated, reserved entries, or restart the lookup if the
+ * xarray was concurrently modified, via xas_retry() ("retry" means the
+ * entry holds an internal xarray value, i.e. is either invalid or NULL
+ * from the caller's perspective).
+ *
+ * Use xas_next() when looking for non-zero attributes to optimize for
+ * the case where the start of the range (or the entire range) doesn't
+ * have any attributes, as xas_next() returns literally the next entry,
+ * whereas xas_next_entry() returns the next non-NULL entry (bounded by
+ * a maximum index).
+ */
for (index = start; index < end; index++) {
do {
- entry = xas_next(&xas);
+ entry = attrs ? xas_next(&xas) :
+ xas_next_entry(&xas, end - 1);
} while (xas_retry(&xas, entry));
+ if (!entry)
+ return !attrs;
+
+ WARN_ON_ONCE(!xa_to_value(entry));
+
if (xas.xa_index != index ||
(xa_to_value(entry) & mask) != attrs)
return false;
@@ -2571,9 +2593,10 @@ static int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
/*
* Reserve memory ahead of time to avoid having to deal with failures
- * partway through setting the new attributes.
+ * partway through setting the new attributes. Storing NULL never
+ * allocates, so no reservations are needed when clearing.
*/
- for (i = start; i < end; i++) {
+ for (i = start; entry && i < end; i++) {
r = xa_reserve(&kvm->mem_attr_array, i, GFP_KERNEL_ACCOUNT);
if (r)
goto out_unlock;