KVM fixes for 7.3-rcN

- Fix a brown paper bag bug where KVM would incorrectly treat Intel PMU MSRs
    as valid on AMD.
 
  - Fix a regression in the hardware disable selftest where it checked the wrong
    macro when detecting glibc support (breaks at least musl).
 
  - Never clear KVM_REQ_VM_DEAD so that dead VMs stay dead, which is especially
    important for KVM_BUG_ON() flows, which often guard more dangerous bugs.
 
  - Re-pend GET_NESTED_STATE_PAGES if getting the pages fails, to fix a bug
    where KVM would let userspace run a broken setup with stale vmcs12 pages.
 
  - Fix a class of bugs where KVM would fail to fill kvm_run exit fields if
    getting nested pages failed.
 
  - Treat reserved entries in the memory attributes xarray as "no attributes",
    to fix false positives when checking for mixed attributes.
 
  - Fix memcg accounting for the memory attributes xarray (the xarray library
    subtly requires the xarray to be configured for accounting upfront; the gfp
    flags taken at runtime are used only rarely).
 
  - Don't pre-reserve xarray entries when storing empty attributes, as storing
    NULL must not require memory allocation (KVM and other subsystems heavily
    rely on this behavior).
 -----BEGIN PGP SIGNATURE-----
 
 iQIzBAABCgAdFiEEKTobbabEP7vbhhN9OlYIJqCjN/0FAmq3A78ACgkQOlYIJqCj
 N/1uVA//TVDznH49htPNV4dh7vfiE8oF9ka4e9KTPA2eMdDeNVInNXBz0ZbJCdNN
 Hm7iZZfk8yx0JCCGXdp8EoYg66iwulBr0KZXRP2U2LoyiRELA1VtGGOLUVP3PXjF
 hElXD56+RrkJJSucINlFU7kSfl2hY/Jbaq87HVBLpXp8iY3stBbLdgMT1ReLdorO
 s9i2uJRqHKjtSG330qDYEG1TtRAdUehgRbxbbJVuOEYtU37tCvX6ai8dEjZAvxYo
 1VZOD05ITo12HUXPiZjIMoNTujQnZKmGYlBYczWlfOmkY/tN6SV8M41srV3ZcK0k
 XGOQM6tOzbVTWaSH/pW1krklnOFAbnL6QzHzb3enHZfUN8AikHPDyfsJWzt/s0f/
 ATTQk03OYL4fJMQSQ778EI6S4PcOzY9HiBxDXM5ySpUv1uyXw8ZFsQSdH/TYTMhi
 Dce42r5rlGa1vAKHwI1Xya2su3REhWzxYxGpUOg/PtRRqWvL2RMrHArzEd6zu+Ti
 jPinuQuXYcT1FTCa5S39NWhU4gWpjw8Pvs9VuUYSCDYxZCaqQ2YheYgsAcgPqgfE
 hS43XVmmOB/tXGsvu+/YuEfbzNV/mHOuJBcjBnnsBP3d6EFB02FPKg4/LYiXujYo
 OocUbSB7bKd87W3B+Ys3QmgOtZh4qX/aLYusa2vQ4Dp9ujXqcWs=
 =LG0B
 -----END PGP SIGNATURE-----

Merge tag 'kvm-x86-fixes-7.3-rc5' of https://github.com/kvm-x86/linux into HEAD

KVM fixes for 7.3-rcN

 - Fix a brown paper bag bug where KVM would incorrectly treat Intel PMU MSRs
   as valid on AMD.

 - Fix a regression in the hardware disable selftest where it checked the wrong
   macro when detecting glibc support (breaks at least musl).

 - Never clear KVM_REQ_VM_DEAD so that dead VMs stay dead, which is especially
   important for KVM_BUG_ON() flows, which often guard more dangerous bugs.

 - Re-pend GET_NESTED_STATE_PAGES if getting the pages fails, to fix a bug
   where KVM would let userspace run a broken setup with stale vmcs12 pages.

 - Fix a class of bugs where KVM would fail to fill kvm_run exit fields if
   getting nested pages failed.

 - Treat reserved entries in the memory attributes xarray as "no attributes",
   to fix false positives when checking for mixed attributes.

 - Fix memcg accounting for the memory attributes xarray (the xarray library
   subtly requires the xarray to be configured for accounting upfront; the gfp
   flags taken at runtime are used only rarely).

 - Don't pre-reserve xarray entries when storing empty attributes, as storing
   NULL must not require memory allocation (KVM and other subsystems heavily
   rely on this behavior).
This commit is contained in:
Paolo Bonzini 2026-09-26 00:40:07 -04:00
commit c2f24f140c
11 changed files with 56 additions and 39 deletions

View File

@ -1133,7 +1133,7 @@ static int kvm_vcpu_suspend(struct kvm_vcpu *vcpu)
static int check_vcpu_requests(struct kvm_vcpu *vcpu)
{
if (kvm_request_pending(vcpu)) {
if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu))
if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu))
return -EIO;
if (kvm_check_request(KVM_REQ_SLEEP, vcpu))

View File

@ -5058,7 +5058,7 @@ static int kvm_tdp_page_prefault(struct kvm_vcpu *vcpu, gpa_t gpa,
if (signal_pending(current))
return -EINTR;
if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu))
if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu))
return -EIO;
cond_resched();

View File

@ -823,14 +823,6 @@ void kvm_pmu_deliver_pmi(struct kvm_vcpu *vcpu)
bool kvm_pmu_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr)
{
switch (msr) {
case MSR_CORE_PERF_GLOBAL_STATUS:
case MSR_CORE_PERF_GLOBAL_CTRL:
case MSR_CORE_PERF_GLOBAL_OVF_CTRL:
return kvm_pmu_has_perf_global_ctrl(vcpu_to_pmu(vcpu));
default:
break;
}
return kvm_pmu_call(msr_idx_to_pmc)(vcpu, msr) ||
kvm_pmu_call(is_valid_msr)(vcpu, msr);
}

View File

@ -2125,13 +2125,8 @@ static bool svm_get_nested_state_pages(struct kvm_vcpu *vcpu)
return false;
}
if (!nested_svm_merge_msrpm(vcpu)) {
vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
vcpu->run->internal.suberror =
KVM_INTERNAL_ERROR_EMULATION;
vcpu->run->internal.ndata = 0;
if (!nested_svm_merge_msrpm(vcpu))
return false;
}
if (kvm_hv_verify_vp_assist(vcpu))
return false;

View File

@ -3465,10 +3465,6 @@ static bool nested_get_vmcs12_pages(struct kvm_vcpu *vcpu)
} else {
pr_debug_ratelimited("%s: no backing for APIC-access address in vmcs12\n",
__func__);
vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
vcpu->run->internal.suberror =
KVM_INTERNAL_ERROR_EMULATION;
vcpu->run->internal.ndata = 0;
return false;
}
}
@ -3539,11 +3535,6 @@ static bool vmx_get_nested_state_pages(struct kvm_vcpu *vcpu)
if (!nested_get_evmcs_page(vcpu)) {
pr_debug_ratelimited("%s: enlightened vmptrld failed\n",
__func__);
vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
vcpu->run->internal.suberror =
KVM_INTERNAL_ERROR_EMULATION;
vcpu->run->internal.ndata = 0;
return false;
}
#endif
@ -3915,8 +3906,12 @@ static int nested_vmx_run(struct kvm_vcpu *vcpu, bool launch)
vmentry_failed:
vcpu->arch.nested_run_pending = 0;
if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR)
if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR) {
vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION;
vcpu->run->internal.ndata = 0;
return 0;
}
if (status == NVMX_VMENTRY_VMEXIT)
return 1;
WARN_ON_ONCE(status != NVMX_VMENTRY_VMFAIL);

View File

@ -187,6 +187,9 @@ static bool intel_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr)
int ret;
switch (msr) {
case MSR_CORE_PERF_GLOBAL_STATUS:
case MSR_CORE_PERF_GLOBAL_CTRL:
case MSR_CORE_PERF_GLOBAL_OVF_CTRL:
case MSR_CORE_PERF_FIXED_CTR_CTRL:
return kvm_pmu_has_perf_global_ctrl(pmu);
case MSR_IA32_PEBS_ENABLE:

View File

@ -1995,7 +1995,7 @@ static int tdx_handle_ept_violation(struct kvm_vcpu *vcpu)
if (kvm_vcpu_has_events(vcpu) || signal_pending(current))
break;
if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) {
if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) {
ret = -EIO;
break;
}

View File

@ -8060,7 +8060,7 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu)
bool req_immediate_exit = false;
if (kvm_request_pending(vcpu)) {
if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) {
if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) {
r = -EIO;
goto out;
}
@ -8072,6 +8072,10 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu)
if (kvm_check_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu)) {
if (unlikely(!kvm_nested_call(get_nested_state_pages)(vcpu))) {
vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR;
vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION;
vcpu->run->internal.ndata = 0;
kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu);
r = 0;
goto out;
}

View File

@ -2324,13 +2324,18 @@ static inline bool kvm_test_request(int req, struct kvm_vcpu *vcpu)
return test_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
static inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
static __always_inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu)
{
BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
clear_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests);
}
static inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
static __always_inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu)
{
/* Once a VM is dead, it needs to stay dead. */
BUILD_BUG_ON(req == KVM_REQ_VM_DEAD);
if (kvm_test_request(req, vcpu)) {
kvm_clear_request(req, vcpu);

View File

@ -37,7 +37,7 @@ static void *run_vcpu(void *arg)
struct kvm_vcpu *vcpu = arg;
struct kvm_run *run = vcpu->run;
#ifndef _GNU_SOURCE
#ifndef __GLIBC__
kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
#endif
@ -51,7 +51,7 @@ static void *sleeping_thread(void *arg)
{
int fd;
#ifndef _GNU_SOURCE
#ifndef __GLIBC__
kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set);
#endif
@ -71,7 +71,7 @@ static void run_test(u32 run)
u32 i, j;
TEST_ASSERT_EQ(pthread_attr_init(&attr), 0);
#ifdef _GNU_SOURCE
#ifdef __GLIBC__
TEST_ASSERT_EQ(pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &threads_cpu_set), 0);
#endif

View File

@ -1117,7 +1117,7 @@ static struct kvm *kvm_create_vm(unsigned long type, const char *fdname)
rcuwait_init(&kvm->mn_memslots_update_rcuwait);
xa_init(&kvm->vcpu_array);
#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES
xa_init(&kvm->mem_attr_array);
xa_init_flags(&kvm->mem_attr_array, XA_FLAGS_ACCOUNT);
#endif
INIT_LIST_HEAD(&kvm->gpc_list);
@ -2447,14 +2447,36 @@ bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
return (kvm_get_memory_attributes(kvm, start) & mask) == attrs;
guard(rcu)();
if (!attrs)
return !xas_find(&xas, end - 1);
/*
* Lookup the entry for each index instead of iterating over the xarray
* as KVM deletes/nullifies entries to represent "no attributes", and
* the xas index is effectively invalid when no entry is found. I.e.
* matching non-zero attributes for *every* entry effectively requires
* a manually lookup for each index.
*
* Skip pre-allocated, reserved entries, or restart the lookup if the
* xarray was concurrently modified, via xas_retry() ("retry" means the
* entry holds an internal xarray value, i.e. is either invalid or NULL
* from the caller's perspective).
*
* Use xas_next() when looking for non-zero attributes to optimize for
* the case where the start of the range (or the entire range) doesn't
* have any attributes, as xas_next() returns literally the next entry,
* whereas xas_next_entry() returns the next non-NULL entry (bounded by
* a maximum index).
*/
for (index = start; index < end; index++) {
do {
entry = xas_next(&xas);
entry = attrs ? xas_next(&xas) :
xas_next_entry(&xas, end - 1);
} while (xas_retry(&xas, entry));
if (!entry)
return !attrs;
WARN_ON_ONCE(!xa_to_value(entry));
if (xas.xa_index != index ||
(xa_to_value(entry) & mask) != attrs)
return false;
@ -2571,9 +2593,10 @@ static int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end,
/*
* Reserve memory ahead of time to avoid having to deal with failures
* partway through setting the new attributes.
* partway through setting the new attributes. Storing NULL never
* allocates, so no reservations are needed when clearing.
*/
for (i = start; i < end; i++) {
for (i = start; entry && i < end; i++) {
r = xa_reserve(&kvm->mem_attr_array, i, GFP_KERNEL_ACCOUNT);
if (r)
goto out_unlock;