KVM SVM changes for 7.3

- Remove a dying VM from the GA Log notifier list before the VM is actually
    destroyed, to fix a potential use-after-free.
 
  - Don't pass FOLL_WRITE when registering encrypted memory regions, i.e. when
    pinning SEV/SEV-ES guest memory, to fix a regression with file-backed memory
    introduced by KVM's (correct) usage of long-term pins.
 
  - Allocate full pages for SEV/SEV-ES {DE,EN}CRYPT ops on SNP-enabled hosts to
    fix a data corruption issue due to the PSP driver assigning to-be-written
    pages to firmware (as required by the SNP specs).
 
  - Unconditionally intercept ICBEP so that KVM generates the correct guest RIP
    when handling an ICEBP-induced TASK_SWITCH #VMEXIT.
 -----BEGIN PGP SIGNATURE-----
 
 iQIzBAABCgAdFiEEKTobbabEP7vbhhN9OlYIJqCjN/0FAmp80WUACgkQOlYIJqCj
 N/3JuhAAv0zwXVVCQbE8Ni3CT/Wl03MEXGDsqlv1mBbsI5WkOte4CshCK8oHAYs4
 LDhO6Y4BlIRJ78itpzk+FF5pvx0MPLi535PNx71o2IGKRZ2T/0u9GVFvHjXdM61F
 OdMFSbRYTg4EHPhqskYN4vIy11Qd3vhaklwcDXxnxlZeTq2wtuVwa1GM7rkmiyRs
 w3dtNrjSTM2axc4+H+gIN/WN5wM5jxsMVxhPdhPzK2isl4+btD9PG7BG64+aZVqJ
 qW9CVTEz81yXdR04sTYutV0Hf56TzGE1RC75MH5nMi7AT9Lfq9L8eVSqPpCZBG/z
 YTO8oSQtD9SrIabE8YQ/pBf+LH4DpDCnxjA5IDeiP/6BW54Via6PXpDQQGGrgqWF
 pWs2jlc1nUxlrUPrPsQxx1yiFk0xDb/ketgwCHvVilyT5CV33ZO3w995FA3Lvjgy
 KY12jxl1jwb8SgZfEkVhUQBNSIsfhj6Wb+jJdbvgcA5M7iUbKk7bBGaokhTbGsev
 e63JCVFRydidINyk+LsvUfgjm1ua18x97A7ihcNMIsJY9CCALK8B/HtyPena2phM
 F+EZ+IydI/JIrXU4C/o0tWm+mlKbGsxAoCcVx/OkIsckEQYhBgcYPlKr5hZvzy1S
 deEmOEXB9cUIO11AOELYnRwEJl6xNuxa226It7WUtmuYQ9t5z48=
 =MkIK
 -----END PGP SIGNATURE-----

Merge tag 'kvm-x86-svm-7.3' of https://github.com/kvm-x86/linux into HEAD

KVM SVM changes for 7.3

 - Remove a dying VM from the GA Log notifier list before the VM is actually
   destroyed, to fix a potential use-after-free.

 - Don't pass FOLL_WRITE when registering encrypted memory regions, i.e. when
   pinning SEV/SEV-ES guest memory, to fix a regression with file-backed memory
   introduced by KVM's (correct) usage of long-term pins.

   [This is correct because, while FOLL_WRITE was needed in the past to trigger
   CoW unsharing, nowadays FOLL_LONGTERM does that already even without
   FOLL_WRITE.  And in fact, get_user_pages() actually disallows FOLL_WRITE
   together with FOLL_LONGTERM.  This change was acked by the MM maintainers.
   For more inforamtion see commit ee1a586dd1.
   - Paolo]

 - Allocate full pages for SEV/SEV-ES {DE,EN}CRYPT ops on SNP-enabled hosts to
   fix a data corruption issue due to the PSP driver assigning to-be-written
   pages to firmware (as required by the SNP specs).

 - Unconditionally intercept ICBEP so that KVM generates the correct guest RIP
   when handling an ICEBP-induced TASK_SWITCH #VMEXIT.
This commit is contained in:
Paolo Bonzini 2026-08-18 13:38:13 +02:00
commit 967872e4f3
5 changed files with 113 additions and 51 deletions

View File

@ -285,13 +285,10 @@ static int avic_get_physical_id_table_order(struct kvm *kvm)
return get_order((__avic_get_max_physical_id(kvm, NULL) + 1) * sizeof(u64));
}
int avic_alloc_physical_id_table(struct kvm *kvm)
static int avic_alloc_physical_id_table(struct kvm *kvm)
{
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
if (!irqchip_in_kernel(kvm) || !enable_apicv)
return 0;
if (kvm_svm->avic_physical_id_table)
return 0;
@ -303,39 +300,32 @@ int avic_alloc_physical_id_table(struct kvm *kvm)
return 0;
}
void avic_vm_destroy(struct kvm *kvm)
static int avic_alloc_logical_id_table(struct kvm *kvm)
{
unsigned long flags;
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
if (!enable_apicv)
return;
free_page((unsigned long)kvm_svm->avic_logical_id_table);
free_pages((unsigned long)kvm_svm->avic_physical_id_table,
avic_get_physical_id_table_order(kvm));
spin_lock_irqsave(&svm_vm_data_hash_lock, flags);
hash_del(&kvm_svm->hnode);
spin_unlock_irqrestore(&svm_vm_data_hash_lock, flags);
}
int avic_vm_init(struct kvm *kvm)
{
unsigned long flags;
int err = -ENOMEM;
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
struct kvm_svm *k2;
u32 vm_id;
if (!enable_apicv)
if (kvm_svm->avic_logical_id_table)
return 0;
kvm_svm->avic_logical_id_table = (void *)get_zeroed_page(GFP_KERNEL_ACCOUNT);
if (!kvm_svm->avic_logical_id_table)
goto free_avic;
return -ENOMEM;
spin_lock_irqsave(&svm_vm_data_hash_lock, flags);
return 0;
}
static void avic_add_vm_to_ga_log_list(struct kvm *kvm)
{
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
struct kvm_svm *k2;
u32 vm_id;
lockdep_assert_held(&kvm->lock);
if (kvm_svm->avic_vm_id)
return;
guard(spinlock_irqsave)(&svm_vm_data_hash_lock);
again:
vm_id = next_vm_id = (next_vm_id + 1) & AVIC_VM_ID_MASK;
if (vm_id == 0) { /* id is 1-based, zero is not okay */
@ -351,13 +341,53 @@ int avic_vm_init(struct kvm *kvm)
}
kvm_svm->avic_vm_id = vm_id;
hash_add(svm_vm_data_hash, &kvm_svm->hnode, kvm_svm->avic_vm_id);
spin_unlock_irqrestore(&svm_vm_data_hash_lock, flags);
}
int avic_vcpu_precreate(struct kvm *kvm)
{
int r;
if (!irqchip_in_kernel(kvm) || WARN_ON_ONCE(!enable_apicv))
return 0;
/*
* Don't unwind on failure, all actions must be idempotent with respect
* to creating multiple vCPUs, i.e. must persist until the VM is destroyed.
*/
r = avic_alloc_physical_id_table(kvm);
if (r)
return r;
r = avic_alloc_logical_id_table(kvm);
if (r)
return r;
avic_add_vm_to_ga_log_list(kvm);
return 0;
}
free_avic:
avic_vm_destroy(kvm);
return err;
void avic_vm_pre_destroy(struct kvm *kvm)
{
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
if (WARN_ON_ONCE(!enable_apicv) || !kvm_svm->avic_vm_id)
return;
guard(spinlock_irqsave)(&svm_vm_data_hash_lock);
hash_del(&kvm_svm->hnode);
}
void avic_vm_destroy(struct kvm *kvm)
{
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
if (!enable_apicv)
return;
free_page((unsigned long)kvm_svm->avic_logical_id_table);
free_pages((unsigned long)kvm_svm->avic_physical_id_table,
avic_get_physical_id_table_order(kvm));
}
static phys_addr_t avic_get_backing_page_address(struct vcpu_svm *svm)

View File

@ -2101,7 +2101,6 @@ static int svm_set_nested_state(struct kvm_vcpu *vcpu,
kvm_make_request(KVM_REQ_APICV_UPDATE, vcpu);
kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu);
ret = 0;
out_free:
kfree(save);
kfree(ctl);

View File

@ -1283,9 +1283,28 @@ static void *sev_dbg_crypt_slow_alloc(struct page *page, unsigned long __va,
if (WARN_ON_ONCE((*pa & PAGE_MASK) != ((*pa + *nr_bytes - 1) & PAGE_MASK)))
return NULL;
/*
* If SNP is enabled, i.e. the RMP is active, allocate a full page to
* prevent concurrent accesses to the page. As required by firmware,
* the PSP driver updates the RMP to temporarily transfer ownership of
* the page to Firmware while the {DE,EN}CRYPT operation is in-progress,
* and so concurrent software accesses to the page will encounter
* seemingly spurious RMP #PF violations
*/
if (cc_platform_has(CC_ATTR_HOST_SEV_SNP))
return (void *)__get_free_page(GFP_KERNEL);
return kmalloc(*nr_bytes, GFP_KERNEL);
}
static void sev_dbg_crypt_slow_free(void *buf)
{
if (cc_platform_has(CC_ATTR_HOST_SEV_SNP))
free_page((unsigned long)buf);
else
kfree(buf);
}
static int sev_dbg_decrypt_slow(struct kvm *kvm, unsigned long src,
struct page *src_p, unsigned long dst,
unsigned int len, int *err)
@ -1307,7 +1326,7 @@ static int sev_dbg_decrypt_slow(struct kvm *kvm, unsigned long src,
if (copy_to_user((void __user *)dst, buf + (src & 15), len))
r = -EFAULT;
out:
kfree(buf);
sev_dbg_crypt_slow_free(buf);
return r;
}
@ -1340,7 +1359,7 @@ static int sev_dbg_encrypt_slow(struct kvm *kvm, unsigned long src,
r = sev_issue_dbg_cmd(kvm, __sme_set(__pa(buf)), dst_pa,
nr_bytes, KVM_SEV_DBG_ENCRYPT, err);
out:
kfree(buf);
sev_dbg_crypt_slow_free(buf);
return r;
}
@ -2757,8 +2776,12 @@ int sev_mem_enc_register_region(struct kvm *kvm,
if (!region)
return -ENOMEM;
/*
* Do NOT specify FOLL_WRITE, as KVM isn't using the pinned pages to
* write memory, and FOLL_LONGTERM itself triggers CoW unshare.
*/
region->pages = sev_pin_memory(kvm, range->addr, range->size, &region->npages,
FOLL_WRITE | FOLL_LONGTERM);
FOLL_LONGTERM);
if (IS_ERR(region->pages)) {
ret = PTR_ERR(region->pages);
goto e_free;

View File

@ -1179,6 +1179,7 @@ static void init_vmcb(struct kvm_vcpu *vcpu, bool init_event)
svm_set_intercept(svm, INTERCEPT_SKINIT);
svm_set_intercept(svm, INTERCEPT_WBINVD);
svm_set_intercept(svm, INTERCEPT_XSETBV);
svm_set_intercept(svm, INTERCEPT_ICEBP);
svm_set_intercept(svm, INTERCEPT_RDPRU);
svm_set_intercept(svm, INTERCEPT_RSM);
@ -1311,11 +1312,6 @@ void svm_switch_vmcb(struct vcpu_svm *svm, struct kvm_vmcb_info *target_vmcb)
svm->vmcb = target_vmcb->ptr;
}
static int svm_vcpu_precreate(struct kvm *kvm)
{
return avic_alloc_physical_id_table(kvm);
}
static int svm_vcpu_create(struct kvm_vcpu *vcpu)
{
struct vcpu_svm *svm;
@ -2081,6 +2077,22 @@ static int bp_interception(struct kvm_vcpu *vcpu)
return 0;
}
static int icebp_interception(struct kvm_vcpu *vcpu)
{
/*
* Intercept and emulate ICEBP (INT1, opcode 0xF1) instead of allowing
* the guest to natively take the #DB trap, so that RIP is advanced
* past the instruction *before* #DB is injected. This is necessary
* because SVM reports the wrong RIP for ICEBP-induced #DB when #DBs
* are delivered via a task gate: RIP points at the ICEBP instruction
* instead of after it (and SVM doesn't provide enough information for
* KVM to detect and manually advance the pre-#DB RIP).
*/
svm_skip_emulated_instruction(vcpu);
kvm_queue_exception(vcpu, DB_VECTOR);
return 1;
}
static int ud_interception(struct kvm_vcpu *vcpu)
{
return handle_ud(vcpu);
@ -3395,6 +3407,7 @@ static int (*const svm_exit_handlers[])(struct kvm_vcpu *vcpu) = {
[SVM_EXIT_MONITOR] = kvm_emulate_monitor,
[SVM_EXIT_MWAIT] = kvm_emulate_mwait,
[SVM_EXIT_XSETBV] = kvm_emulate_xsetbv,
[SVM_EXIT_ICEBP] = icebp_interception,
[SVM_EXIT_RDPRU] = kvm_handle_invalid_op,
[SVM_EXIT_EFER_WRITE_TRAP] = efer_trap,
[SVM_EXIT_CR0_WRITE_TRAP] = cr_trap,
@ -5320,12 +5333,6 @@ static int svm_vm_init(struct kvm *kvm)
if (!pause_filter_count || !pause_filter_thresh)
kvm_disable_exits(kvm, KVM_X86_DISABLE_EXITS_PAUSE);
if (enable_apicv) {
int ret = avic_vm_init(kvm);
if (ret)
return ret;
}
svm_srso_vm_init();
return 0;
}
@ -5351,13 +5358,14 @@ struct kvm_x86_ops svm_x86_ops __initdata = {
.emergency_disable_virtualization_cpu = svm_emergency_disable_virtualization_cpu,
.has_emulated_msr = svm_has_emulated_msr,
.vcpu_precreate = svm_vcpu_precreate,
.vcpu_precreate = avic_vcpu_precreate,
.vcpu_create = svm_vcpu_create,
.vcpu_free = svm_vcpu_free,
.vcpu_reset = svm_vcpu_reset,
.vm_size = sizeof(struct kvm_svm),
.vm_init = svm_vm_init,
.vm_pre_destroy = avic_vm_pre_destroy,
.vm_destroy = svm_vm_destroy,
.prepare_switch_to_guest = svm_prepare_switch_to_guest,
@ -5732,6 +5740,8 @@ static __init int svm_hardware_setup(void)
enable_apicv = avic_hardware_setup();
if (!enable_apicv) {
enable_ipiv = false;
svm_x86_ops.vcpu_precreate = NULL;
svm_x86_ops.vm_pre_destroy = NULL;
svm_x86_ops.vcpu_blocking = NULL;
svm_x86_ops.vcpu_unblocking = NULL;
svm_x86_ops.vcpu_get_apicv_inhibit_reasons = NULL;

View File

@ -947,9 +947,9 @@ extern struct kvm_x86_nested_ops svm_nested_ops;
bool __init avic_hardware_setup(void);
void avic_hardware_unsetup(void);
int avic_alloc_physical_id_table(struct kvm *kvm);
int avic_vcpu_precreate(struct kvm *kvm);
void avic_vm_pre_destroy(struct kvm *kvm);
void avic_vm_destroy(struct kvm *kvm);
int avic_vm_init(struct kvm *kvm);
void avic_init_vmcb(struct vcpu_svm *svm, struct vmcb *vmcb);
int avic_incomplete_ipi_interception(struct kvm_vcpu *vcpu);
int avic_unaccelerated_access_interception(struct kvm_vcpu *vcpu);