From 79a71cc2568f4b5d42284da2aa26f3b4f47ce01b Mon Sep 17 00:00:00 2001 From: Jim Mattson Date: Wed, 2 Sep 2026 11:47:11 -0700 Subject: [PATCH 1/8] KVM: x86/pmu: Move Intel PMU global MSRs to intel_is_valid_msr() Commit c85cdc1cc1ea ("KVM: x86/pmu: Move handling PERF_GLOBAL_CTRL and friends to common x86") moved the existence check for the following Intel PMU MSRs to kvm_pmu_is_valid_msr(): - MSR_CORE_PERF_GLOBAL_STATUS - MSR_CORE_PERF_GLOBAL_CTRL - MSR_CORE_PERF_GLOBAL_OVF_CTRL That commit deemed these MSRs valid whenever pmu->version > 1. It intended to share the check with AMD PerfMonV2 because both vendor implementations require version 2 or greater for global PMU controls. However, as noted in the commit message, AMD uses different MSR indices for its global PMU registers. Commit 4a2771895ca6 ("KVM: x86/svm/pmu: Add AMD PerfMonV2 support") subsequently added AMD PerfMonV2 support and set pmu->version = 2. Because kvm_pmu_is_valid_msr() validated the Intel MSRs whenever pmu->version > 1, KVM incorrectly permitted AMD guests with PerfMonV2 to access these Intel MSRs without a #GP. Move the validation of these Intel MSRs to intel_is_valid_msr() and remove the common switch statement from kvm_pmu_is_valid_msr(). AMD already validates its own global PMU MSRs in amd_is_valid_msr(). Fixes: 4a2771895ca6 ("KVM: x86/svm/pmu: Add AMD PerfMonV2 support") Signed-off-by: Jim Mattson Reviewed-by: Like Xu Reviewed-by: Sandipan Das Link: https://patch.msgid.link/20260902184711.138538-1-jmattson@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/pmu.c | 8 -------- arch/x86/kvm/vmx/pmu_intel.c | 3 +++ 2 files changed, 3 insertions(+), 8 deletions(-) diff --git a/arch/x86/kvm/pmu.c b/arch/x86/kvm/pmu.c index a7d60c8785cd..d2fd47ee5ec8 100644 --- a/arch/x86/kvm/pmu.c +++ b/arch/x86/kvm/pmu.c @@ -823,14 +823,6 @@ void kvm_pmu_deliver_pmi(struct kvm_vcpu *vcpu) bool kvm_pmu_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr) { - switch (msr) { - case MSR_CORE_PERF_GLOBAL_STATUS: - case MSR_CORE_PERF_GLOBAL_CTRL: - case MSR_CORE_PERF_GLOBAL_OVF_CTRL: - return kvm_pmu_has_perf_global_ctrl(vcpu_to_pmu(vcpu)); - default: - break; - } return kvm_pmu_call(msr_idx_to_pmc)(vcpu, msr) || kvm_pmu_call(is_valid_msr)(vcpu, msr); } diff --git a/arch/x86/kvm/vmx/pmu_intel.c b/arch/x86/kvm/vmx/pmu_intel.c index bfa8612fb450..70a8c4816135 100644 --- a/arch/x86/kvm/vmx/pmu_intel.c +++ b/arch/x86/kvm/vmx/pmu_intel.c @@ -187,6 +187,9 @@ static bool intel_is_valid_msr(struct kvm_vcpu *vcpu, u32 msr) int ret; switch (msr) { + case MSR_CORE_PERF_GLOBAL_STATUS: + case MSR_CORE_PERF_GLOBAL_CTRL: + case MSR_CORE_PERF_GLOBAL_OVF_CTRL: case MSR_CORE_PERF_FIXED_CTR_CTRL: return kvm_pmu_has_perf_global_ctrl(pmu); case MSR_IA32_PEBS_ENABLE: From f13368e0acffd6d6289629705cb821d921104725 Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Wed, 26 Aug 2026 09:14:30 -0700 Subject: [PATCH 2/8] KVM: selftests: Use __GLIBC__, not _GNU_SOURCE, to detect actual glibc Use __GLIBC__ in the hardware disable test to detect when selftests are being built/linked against glibc and thus pthread_attr_setaffinity_np() is (hopefully) available. As pointed out by Sashiko and Hisam, _GNU_SOURCE is effectively a "request" macro to enable functionality, whereas __GLIBC__ is an announcement of support and selftests' idiomatic way of guarding code that's specific to glibc. Fixes: 496779b54943 ("KVM: selftests: Pre-set threads affinity in hardware disable test when possible") Reported-by: Sashiko Bot Closes: https://lore.kernel.org/all/20260731201140.5AF0C1F00AC4@smtp.kernel.org Suggested-by: Hisam Mehboob Link: https://patch.msgid.link/20260826161430.714316-1-seanjc@google.com Signed-off-by: Sean Christopherson --- tools/testing/selftests/kvm/hardware_disable_test.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tools/testing/selftests/kvm/hardware_disable_test.c b/tools/testing/selftests/kvm/hardware_disable_test.c index 43a36ef3ead8..1c20892d6782 100644 --- a/tools/testing/selftests/kvm/hardware_disable_test.c +++ b/tools/testing/selftests/kvm/hardware_disable_test.c @@ -37,7 +37,7 @@ static void *run_vcpu(void *arg) struct kvm_vcpu *vcpu = arg; struct kvm_run *run = vcpu->run; -#ifndef _GNU_SOURCE +#ifndef __GLIBC__ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set); #endif @@ -51,7 +51,7 @@ static void *sleeping_thread(void *arg) { int fd; -#ifndef _GNU_SOURCE +#ifndef __GLIBC__ kvm_sched_setaffinity(0, sizeof(cpu_set_t), &threads_cpu_set); #endif @@ -71,7 +71,7 @@ static void run_test(u32 run) u32 i, j; TEST_ASSERT_EQ(pthread_attr_init(&attr), 0); -#ifdef _GNU_SOURCE +#ifdef __GLIBC__ TEST_ASSERT_EQ(pthread_attr_setaffinity_np(&attr, sizeof(cpu_set_t), &threads_cpu_set), 0); #endif From 8cd280282d1239bca68f8c3632ef1ac8556a396b Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Thu, 6 Aug 2026 14:46:18 -0700 Subject: [PATCH 3/8] KVM: Never clear KVM_REQ_VM_DEAD from a vCPU's requests Use kvm_test_request() instead of kvm_check_request() when querying KVM_REQ_VM_DEAD, i.e. don't clear KVM_REQ_VM_DEAD, as the entire purpose of KVM_REQ_VM_DEAD is to prevent the vCPU from enterring the guest ever again, even if userspace insists on redoing KVM_RUN. Ensuring KVM_REQ_VM_DEAD is never cleared will allow relaxing KVM's rule that ioctls can't be invoked on dead VMs, to only disallow ioctls if the VM is bugged, i.e. if KVM hit a KVM_BUG_ON(). Opportunistically add compile-time assertions to guard against clearing KVM_REQ_VM_DEAD through the standard APIs. Reviewed-by: Kai Huang Acked-by: Marc Zyngier Link: https://patch.msgid.link/20260806214618.82180-1-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/arm64/kvm/arm.c | 2 +- arch/x86/kvm/mmu/mmu.c | 2 +- arch/x86/kvm/vmx/tdx.c | 2 +- arch/x86/kvm/x86.c | 2 +- include/linux/kvm_host.h | 9 +++++++-- 5 files changed, 11 insertions(+), 6 deletions(-) diff --git a/arch/arm64/kvm/arm.c b/arch/arm64/kvm/arm.c index 8b080804bc90..62c81d6ae8ee 100644 --- a/arch/arm64/kvm/arm.c +++ b/arch/arm64/kvm/arm.c @@ -1129,7 +1129,7 @@ static int kvm_vcpu_suspend(struct kvm_vcpu *vcpu) static int check_vcpu_requests(struct kvm_vcpu *vcpu) { if (kvm_request_pending(vcpu)) { - if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) + if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) return -EIO; if (kvm_check_request(KVM_REQ_SLEEP, vcpu)) diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index 064ecc33b926..8e62476e477b 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -5058,7 +5058,7 @@ static int kvm_tdp_page_prefault(struct kvm_vcpu *vcpu, gpa_t gpa, if (signal_pending(current)) return -EINTR; - if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) + if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) return -EIO; cond_resched(); diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index b272c20586a7..6c842e9191a5 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -1995,7 +1995,7 @@ static int tdx_handle_ept_violation(struct kvm_vcpu *vcpu) if (kvm_vcpu_has_events(vcpu) || signal_pending(current)) break; - if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) { + if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) { ret = -EIO; break; } diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index 79468ddfe473..a137dc6dd8c6 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -8060,7 +8060,7 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) bool req_immediate_exit = false; if (kvm_request_pending(vcpu)) { - if (kvm_check_request(KVM_REQ_VM_DEAD, vcpu)) { + if (kvm_test_request(KVM_REQ_VM_DEAD, vcpu)) { r = -EIO; goto out; } diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index 03bfc92864b6..cf7fe835c4ad 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -2324,13 +2324,18 @@ static inline bool kvm_test_request(int req, struct kvm_vcpu *vcpu) return test_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests); } -static inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu) +static __always_inline void kvm_clear_request(int req, struct kvm_vcpu *vcpu) { + BUILD_BUG_ON(req == KVM_REQ_VM_DEAD); + clear_bit(req & KVM_REQUEST_MASK, (void *)&vcpu->requests); } -static inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu) +static __always_inline bool kvm_check_request(int req, struct kvm_vcpu *vcpu) { + /* Once a VM is dead, it needs to stay dead. */ + BUILD_BUG_ON(req == KVM_REQ_VM_DEAD); + if (kvm_test_request(req, vcpu)) { kvm_clear_request(req, vcpu); From 10180a277549339020b08000206092c07e0bff5a Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Mon, 21 Sep 2026 14:16:07 -0700 Subject: [PATCH 4/8] KVM: x86: Re-pend GET_NESTED_STATE_PAGES if getting said pages fails Re-pend GET_NESTED_STATE_PAGES before exiting to userspace if getting the nested pages fails in the KVM_RUN path. If userspace re-runs the vCPU, and vmcs02 holds valid PFNs from the *previous* run of L2, then KVM could re-enter L2 with stale, unpinned PFNs mapped into e.g. the vAPIC page. Note, both SVM and VMX (as of commit 11722439fb20 ("KVM: nVMX: Ensure KVM_REQ_GET_NESTED_STATE_PAGES is cleared on VM-Exit") ensure the request is cleared on VM-Exit (including the "forced" case), i.e. there is no risk of double-mapping due to emulated VMLAUNCH/VMRESUME/VMRUN *and* the request trying to map the nested pages. Fixes: 671ddc700fd0 ("KVM: nVMX: Don't leak L1 MMIO regions to L2") Cc: stable@vger.kernel.org Reported-by: Jinwoo Lee Closes: https://lore.kernel.org/all/20260813043932.3214460-1-rkskek9254@gmail.com Reported-by: Stefan Teodorescu Link: https://patch.msgid.link/20260921211608.1030158-2-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/x86.c | 1 + 1 file changed, 1 insertion(+) diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index a137dc6dd8c6..4ec17aaff413 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -8072,6 +8072,7 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) if (kvm_check_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu)) { if (unlikely(!kvm_nested_call(get_nested_state_pages)(vcpu))) { + kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu); r = 0; goto out; } From c1214f293d77c6e425a379f90d09bdb4815993a8 Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Mon, 21 Sep 2026 14:16:08 -0700 Subject: [PATCH 5/8] KVM: x86: Fill kvm_run exit fields in common get_nested_state_pages() error paths Fill kvm_run with "internal error, emulation" in the common error handling paths for getting nested state pages, as requiring each check to manually fill kvm_run is error prone and requires a non-trivial amount of copy+paste. Specifically, both SVM and VMX fail to fill kvm_run if load_pdptrs() fails, and SVM fails to fill kvm_run if kvm_hv_verify_vp_assist() fails. If those flows fail, the *best* case scenario is that KVM will exit to userspace with KVM_EXIT_UNKNOWN. The worst case scenario is that KVM exits with a stale exit_reason and confuses userspace. Note, SVM never exits to userspace if something goes sideways when dealing with vmcb12 assets while emulating VMRUN, i.e. lack of SVM-specific code is not a bug. Fixes: 0f85722341b0 ("KVM: nVMX: delay loading of PDPTRs to KVM_REQ_GET_NESTED_STATE_PAGES") Fixes: 232f75d3b4b5 ("KVM: nSVM: call nested_svm_load_cr3 on nested state load") Fixes: 3f4a812edf5c ("KVM: nSVM: hyper-v: Enable L2 TLB flush") Cc: stable@vger.kernel.org Reported-by: Jinwoo Lee Closes: https://lore.kernel.org/all/20260813043932.3214460-1-rkskek9254@gmail.com Reported-by: Stefan Teodorescu Link: https://patch.msgid.link/20260921211608.1030158-3-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/svm/nested.c | 7 +------ arch/x86/kvm/vmx/nested.c | 15 +++++---------- arch/x86/kvm/x86.c | 3 +++ 3 files changed, 9 insertions(+), 16 deletions(-) diff --git a/arch/x86/kvm/svm/nested.c b/arch/x86/kvm/svm/nested.c index 73f37b050d0a..f9090b601efa 100644 --- a/arch/x86/kvm/svm/nested.c +++ b/arch/x86/kvm/svm/nested.c @@ -2125,13 +2125,8 @@ static bool svm_get_nested_state_pages(struct kvm_vcpu *vcpu) return false; } - if (!nested_svm_merge_msrpm(vcpu)) { - vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR; - vcpu->run->internal.suberror = - KVM_INTERNAL_ERROR_EMULATION; - vcpu->run->internal.ndata = 0; + if (!nested_svm_merge_msrpm(vcpu)) return false; - } if (kvm_hv_verify_vp_assist(vcpu)) return false; diff --git a/arch/x86/kvm/vmx/nested.c b/arch/x86/kvm/vmx/nested.c index 151873407abd..40c1a5f6fa8a 100644 --- a/arch/x86/kvm/vmx/nested.c +++ b/arch/x86/kvm/vmx/nested.c @@ -3465,10 +3465,6 @@ static bool nested_get_vmcs12_pages(struct kvm_vcpu *vcpu) } else { pr_debug_ratelimited("%s: no backing for APIC-access address in vmcs12\n", __func__); - vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR; - vcpu->run->internal.suberror = - KVM_INTERNAL_ERROR_EMULATION; - vcpu->run->internal.ndata = 0; return false; } } @@ -3539,11 +3535,6 @@ static bool vmx_get_nested_state_pages(struct kvm_vcpu *vcpu) if (!nested_get_evmcs_page(vcpu)) { pr_debug_ratelimited("%s: enlightened vmptrld failed\n", __func__); - vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR; - vcpu->run->internal.suberror = - KVM_INTERNAL_ERROR_EMULATION; - vcpu->run->internal.ndata = 0; - return false; } #endif @@ -3915,8 +3906,12 @@ static int nested_vmx_run(struct kvm_vcpu *vcpu, bool launch) vmentry_failed: vcpu->arch.nested_run_pending = 0; - if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR) + if (status == NVMX_VMENTRY_KVM_INTERNAL_ERROR) { + vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR; + vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION; + vcpu->run->internal.ndata = 0; return 0; + } if (status == NVMX_VMENTRY_VMEXIT) return 1; WARN_ON_ONCE(status != NVMX_VMENTRY_VMFAIL); diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index 4ec17aaff413..aad065d035fb 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -8072,6 +8072,9 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) if (kvm_check_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu)) { if (unlikely(!kvm_nested_call(get_nested_state_pages)(vcpu))) { + vcpu->run->exit_reason = KVM_EXIT_INTERNAL_ERROR; + vcpu->run->internal.suberror = KVM_INTERNAL_ERROR_EMULATION; + vcpu->run->internal.ndata = 0; kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu); r = 0; goto out; From 277d3623d99a4fc2623bfb7d191b649ca380605d Mon Sep 17 00:00:00 2001 From: Zeng Chi Date: Mon, 21 Sep 2026 18:24:42 +0800 Subject: [PATCH 6/8] KVM: Don't treat reserved xarray entries as having memory attributes kvm_vm_set_mem_attributes() reserves an xarray entry for every gfn in the range before storing the new attributes, so that the store loop can't fail partway through. If one of the reservations fails, e.g. with -ENOMEM, the entries that were already reserved are left in the array. That is harmless as far as xa_reserve() is concerned, as the reserved entries read back as NULL via xa_load(), but it confuses the "does this range have no attributes at all" check: if (!attrs) return !xas_find(&xas, end - 1); A reserved entry is XA_ZERO_ENTRY, not NULL, and xas_find() returns it as present. So a leftover reservation makes KVM report that a fully shared range has attributes even though kvm_get_memory_attributes() returns none for every gfn in the range. On x86, the next time mixed-attribute tracking is recomputed for the range (memslot creation, or a later attribute change that straddles the 2MiB page), hugepage_has_attrs() treats a fully shared 2MiB range as mixed and refuses to map it with a hugepage, until userspace happens to set attributes on the range again. Drop the shortcut and handle the !attrs case in the per-index loop, using xas_next_entry() to find the next non-NULL entry. xas_next_entry() is essentially an optimized xas_find(), so the effective change is that the !attrs lookup now goes through xas_retry() like the attrs != 0 case, i.e. reserved entries are skipped and retry entries restart the walk. Don't check the index when no entry is found, as the xarray leaves the xas index in a bogus state in that case; no entry simply means the rest of the range has no attributes. KVM never stores a non-NULL entry with a value of zero (clearing stores NULL), but such an entry would be returned by xas_next_entry() and trip the index check, so WARN if one is ever seen. Fixes: 5a475554db1e ("KVM: Introduce per-page memory attributes") Cc: stable@vger.kernel.org Suggested-by: Sean Christopherson Cc: David Ballesteros Signed-off-by: Zeng Chi Link: https://patch.msgid.link/20260921102442.1232375-1-zeng_chi911@163.com [sean: expand comment to elaborate on xarray APIs, split optimization out] Signed-off-by: Sean Christopherson --- virt/kvm/kvm_main.c | 28 +++++++++++++++++++++++++--- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 65eb26a0520d..2a046e95f95e 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -2447,14 +2447,36 @@ bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t end, return (kvm_get_memory_attributes(kvm, start) & mask) == attrs; guard(rcu)(); - if (!attrs) - return !xas_find(&xas, end - 1); + /* + * Lookup the entry for each index instead of iterating over the xarray + * as KVM deletes/nullifies entries to represent "no attributes", and + * the xas index is effectively invalid when no entry is found. I.e. + * matching non-zero attributes for *every* entry effectively requires + * a manually lookup for each index. + * + * Skip pre-allocated, reserved entries, or restart the lookup if the + * xarray was concurrently modified, via xas_retry() ("retry" means the + * entry holds an internal xarray value, i.e. is either invalid or NULL + * from the caller's perspective). + * + * Use xas_next() when looking for non-zero attributes to optimize for + * the case where the start of the range (or the entire range) doesn't + * have any attributes, as xas_next() returns literally the next entry, + * whereas xas_next_entry() returns the next non-NULL entry (bounded by + * a maximum index). + */ for (index = start; index < end; index++) { do { - entry = xas_next(&xas); + entry = attrs ? xas_next(&xas) : + xas_next_entry(&xas, end - 1); } while (xas_retry(&xas, entry)); + if (!entry) + return !attrs; + + WARN_ON_ONCE(!xa_to_value(entry)); + if (xas.xa_index != index || (xa_to_value(entry) & mask) != attrs) return false; From 382e5d514b6f35bdda2ab9044b4eed23d2ec4254 Mon Sep 17 00:00:00 2001 From: David Ballesteros Date: Tue, 15 Sep 2026 17:53:57 +0000 Subject: [PATCH 7/8] KVM: Ensure memory attributes xarray nodes are accounted to the caller's memcg Explicitly instantiate the memory attributes xarray with XA_FLAGS_ACCOUNT to ensure that all allocations are accounted to the memcg. Frustratingly, memory allocations done in the "fastpath" do not honor the passed in gfp, even for an explicit xa_reserve(). Only the rare, slow path __xas_nomem() honors the original gfp. E.g. xa_reserve(..., GFP_KERNEL_ACCOUNT) | -> ... | -> __xa_cmpxchg_raw() | -> xas_store() <== does not take @gfp | -> xas_create() | -> xas_alloc() The bug was confirmed by observing that a process in a cgroup limited to 256 MiB grew radix_tree_node slab by ~512 MiB while its memory.current stayed near 0. Fixes: 5a475554db1e ("KVM: Introduce per-page memory attributes") Cc: stable@vger.kernel.org Assisted-by: Claude-Code:claude-opus-5 Signed-off-by: David Ballesteros Link: https://patch.msgid.link/20260915175335.138547-4-davimaba.v@proton.me [sean: rewrite changelog, tag for stable] Signed-off-by: Sean Christopherson --- virt/kvm/kvm_main.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 2a046e95f95e..df643ec3ea75 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -1117,7 +1117,7 @@ static struct kvm *kvm_create_vm(unsigned long type, const char *fdname) rcuwait_init(&kvm->mn_memslots_update_rcuwait); xa_init(&kvm->vcpu_array); #ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES - xa_init(&kvm->mem_attr_array); + xa_init_flags(&kvm->mem_attr_array, XA_FLAGS_ACCOUNT); #endif INIT_LIST_HEAD(&kvm->gpc_list); From 119e9db233ad0f4cbd4cfb9f9385a7bca378b37d Mon Sep 17 00:00:00 2001 From: Zeng Chi Date: Mon, 21 Sep 2026 18:24:42 +0800 Subject: [PATCH 8/8] KVM: Don't pre-reserve xarray entries when storing empty/NULL attributes Skip the xarray reservation loop when clearing all memory attributes, as storing NULL only erases the entry and never needs to allocate, so no reservation (and no cleanup of a failed one) is required in that case. Suggested-by: Sean Christopherson Cc: David Ballesteros Signed-off-by: Zeng Chi Link: https://patch.msgid.link/20260921102442.1232375-1-zeng_chi911@163.com [sean: split to separate patch] Signed-off-by: Sean Christopherson --- virt/kvm/kvm_main.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index df643ec3ea75..85f42289748d 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -2593,9 +2593,10 @@ static int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end, /* * Reserve memory ahead of time to avoid having to deal with failures - * partway through setting the new attributes. + * partway through setting the new attributes. Storing NULL never + * allocates, so no reservations are needed when clearing. */ - for (i = start; i < end; i++) { + for (i = start; entry && i < end; i++) { r = xa_reserve(&kvm->mem_attr_array, i, GFP_KERNEL_ACCOUNT); if (r) goto out_unlock;