mirror of
https://github.com/torvalds/linux.git
synced 2026-09-29 12:24:02 +02:00
KVM SVM changes for 7.3
- Remove a dying VM from the GA Log notifier list before the VM is actually
destroyed, to fix a potential use-after-free.
- Don't pass FOLL_WRITE when registering encrypted memory regions, i.e. when
pinning SEV/SEV-ES guest memory, to fix a regression with file-backed memory
introduced by KVM's (correct) usage of long-term pins.
- Allocate full pages for SEV/SEV-ES {DE,EN}CRYPT ops on SNP-enabled hosts to
fix a data corruption issue due to the PSP driver assigning to-be-written
pages to firmware (as required by the SNP specs).
- Unconditionally intercept ICBEP so that KVM generates the correct guest RIP
when handling an ICEBP-induced TASK_SWITCH #VMEXIT.
-----BEGIN PGP SIGNATURE-----
iQIzBAABCgAdFiEEKTobbabEP7vbhhN9OlYIJqCjN/0FAmp80WUACgkQOlYIJqCj
N/3JuhAAv0zwXVVCQbE8Ni3CT/Wl03MEXGDsqlv1mBbsI5WkOte4CshCK8oHAYs4
LDhO6Y4BlIRJ78itpzk+FF5pvx0MPLi535PNx71o2IGKRZ2T/0u9GVFvHjXdM61F
OdMFSbRYTg4EHPhqskYN4vIy11Qd3vhaklwcDXxnxlZeTq2wtuVwa1GM7rkmiyRs
w3dtNrjSTM2axc4+H+gIN/WN5wM5jxsMVxhPdhPzK2isl4+btD9PG7BG64+aZVqJ
qW9CVTEz81yXdR04sTYutV0Hf56TzGE1RC75MH5nMi7AT9Lfq9L8eVSqPpCZBG/z
YTO8oSQtD9SrIabE8YQ/pBf+LH4DpDCnxjA5IDeiP/6BW54Via6PXpDQQGGrgqWF
pWs2jlc1nUxlrUPrPsQxx1yiFk0xDb/ketgwCHvVilyT5CV33ZO3w995FA3Lvjgy
KY12jxl1jwb8SgZfEkVhUQBNSIsfhj6Wb+jJdbvgcA5M7iUbKk7bBGaokhTbGsev
e63JCVFRydidINyk+LsvUfgjm1ua18x97A7ihcNMIsJY9CCALK8B/HtyPena2phM
F+EZ+IydI/JIrXU4C/o0tWm+mlKbGsxAoCcVx/OkIsckEQYhBgcYPlKr5hZvzy1S
deEmOEXB9cUIO11AOELYnRwEJl6xNuxa226It7WUtmuYQ9t5z48=
=MkIK
-----END PGP SIGNATURE-----
Merge tag 'kvm-x86-svm-7.3' of https://github.com/kvm-x86/linux into HEAD
KVM SVM changes for 7.3
- Remove a dying VM from the GA Log notifier list before the VM is actually
destroyed, to fix a potential use-after-free.
- Don't pass FOLL_WRITE when registering encrypted memory regions, i.e. when
pinning SEV/SEV-ES guest memory, to fix a regression with file-backed memory
introduced by KVM's (correct) usage of long-term pins.
[This is correct because, while FOLL_WRITE was needed in the past to trigger
CoW unsharing, nowadays FOLL_LONGTERM does that already even without
FOLL_WRITE. And in fact, get_user_pages() actually disallows FOLL_WRITE
together with FOLL_LONGTERM. This change was acked by the MM maintainers.
For more inforamtion see commit ee1a586dd1.
- Paolo]
- Allocate full pages for SEV/SEV-ES {DE,EN}CRYPT ops on SNP-enabled hosts to
fix a data corruption issue due to the PSP driver assigning to-be-written
pages to firmware (as required by the SNP specs).
- Unconditionally intercept ICBEP so that KVM generates the correct guest RIP
when handling an ICEBP-induced TASK_SWITCH #VMEXIT.
This commit is contained in:
commit
967872e4f3
|
|
@ -285,13 +285,10 @@ static int avic_get_physical_id_table_order(struct kvm *kvm)
|
|||
return get_order((__avic_get_max_physical_id(kvm, NULL) + 1) * sizeof(u64));
|
||||
}
|
||||
|
||||
int avic_alloc_physical_id_table(struct kvm *kvm)
|
||||
static int avic_alloc_physical_id_table(struct kvm *kvm)
|
||||
{
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
|
||||
if (!irqchip_in_kernel(kvm) || !enable_apicv)
|
||||
return 0;
|
||||
|
||||
if (kvm_svm->avic_physical_id_table)
|
||||
return 0;
|
||||
|
||||
|
|
@ -303,39 +300,32 @@ int avic_alloc_physical_id_table(struct kvm *kvm)
|
|||
return 0;
|
||||
}
|
||||
|
||||
void avic_vm_destroy(struct kvm *kvm)
|
||||
static int avic_alloc_logical_id_table(struct kvm *kvm)
|
||||
{
|
||||
unsigned long flags;
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
|
||||
if (!enable_apicv)
|
||||
return;
|
||||
|
||||
free_page((unsigned long)kvm_svm->avic_logical_id_table);
|
||||
free_pages((unsigned long)kvm_svm->avic_physical_id_table,
|
||||
avic_get_physical_id_table_order(kvm));
|
||||
|
||||
spin_lock_irqsave(&svm_vm_data_hash_lock, flags);
|
||||
hash_del(&kvm_svm->hnode);
|
||||
spin_unlock_irqrestore(&svm_vm_data_hash_lock, flags);
|
||||
}
|
||||
|
||||
int avic_vm_init(struct kvm *kvm)
|
||||
{
|
||||
unsigned long flags;
|
||||
int err = -ENOMEM;
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
struct kvm_svm *k2;
|
||||
u32 vm_id;
|
||||
|
||||
if (!enable_apicv)
|
||||
if (kvm_svm->avic_logical_id_table)
|
||||
return 0;
|
||||
|
||||
kvm_svm->avic_logical_id_table = (void *)get_zeroed_page(GFP_KERNEL_ACCOUNT);
|
||||
if (!kvm_svm->avic_logical_id_table)
|
||||
goto free_avic;
|
||||
return -ENOMEM;
|
||||
|
||||
spin_lock_irqsave(&svm_vm_data_hash_lock, flags);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void avic_add_vm_to_ga_log_list(struct kvm *kvm)
|
||||
{
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
struct kvm_svm *k2;
|
||||
u32 vm_id;
|
||||
|
||||
lockdep_assert_held(&kvm->lock);
|
||||
|
||||
if (kvm_svm->avic_vm_id)
|
||||
return;
|
||||
|
||||
guard(spinlock_irqsave)(&svm_vm_data_hash_lock);
|
||||
again:
|
||||
vm_id = next_vm_id = (next_vm_id + 1) & AVIC_VM_ID_MASK;
|
||||
if (vm_id == 0) { /* id is 1-based, zero is not okay */
|
||||
|
|
@ -351,13 +341,53 @@ int avic_vm_init(struct kvm *kvm)
|
|||
}
|
||||
kvm_svm->avic_vm_id = vm_id;
|
||||
hash_add(svm_vm_data_hash, &kvm_svm->hnode, kvm_svm->avic_vm_id);
|
||||
spin_unlock_irqrestore(&svm_vm_data_hash_lock, flags);
|
||||
}
|
||||
|
||||
int avic_vcpu_precreate(struct kvm *kvm)
|
||||
{
|
||||
int r;
|
||||
|
||||
if (!irqchip_in_kernel(kvm) || WARN_ON_ONCE(!enable_apicv))
|
||||
return 0;
|
||||
|
||||
/*
|
||||
* Don't unwind on failure, all actions must be idempotent with respect
|
||||
* to creating multiple vCPUs, i.e. must persist until the VM is destroyed.
|
||||
*/
|
||||
r = avic_alloc_physical_id_table(kvm);
|
||||
if (r)
|
||||
return r;
|
||||
|
||||
r = avic_alloc_logical_id_table(kvm);
|
||||
if (r)
|
||||
return r;
|
||||
|
||||
avic_add_vm_to_ga_log_list(kvm);
|
||||
return 0;
|
||||
}
|
||||
|
||||
free_avic:
|
||||
avic_vm_destroy(kvm);
|
||||
return err;
|
||||
void avic_vm_pre_destroy(struct kvm *kvm)
|
||||
{
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
|
||||
if (WARN_ON_ONCE(!enable_apicv) || !kvm_svm->avic_vm_id)
|
||||
return;
|
||||
|
||||
guard(spinlock_irqsave)(&svm_vm_data_hash_lock);
|
||||
|
||||
hash_del(&kvm_svm->hnode);
|
||||
}
|
||||
|
||||
void avic_vm_destroy(struct kvm *kvm)
|
||||
{
|
||||
struct kvm_svm *kvm_svm = to_kvm_svm(kvm);
|
||||
|
||||
if (!enable_apicv)
|
||||
return;
|
||||
|
||||
free_page((unsigned long)kvm_svm->avic_logical_id_table);
|
||||
free_pages((unsigned long)kvm_svm->avic_physical_id_table,
|
||||
avic_get_physical_id_table_order(kvm));
|
||||
}
|
||||
|
||||
static phys_addr_t avic_get_backing_page_address(struct vcpu_svm *svm)
|
||||
|
|
|
|||
|
|
@ -2101,7 +2101,6 @@ static int svm_set_nested_state(struct kvm_vcpu *vcpu,
|
|||
kvm_make_request(KVM_REQ_APICV_UPDATE, vcpu);
|
||||
|
||||
kvm_make_request(KVM_REQ_GET_NESTED_STATE_PAGES, vcpu);
|
||||
ret = 0;
|
||||
out_free:
|
||||
kfree(save);
|
||||
kfree(ctl);
|
||||
|
|
|
|||
|
|
@ -1283,9 +1283,28 @@ static void *sev_dbg_crypt_slow_alloc(struct page *page, unsigned long __va,
|
|||
if (WARN_ON_ONCE((*pa & PAGE_MASK) != ((*pa + *nr_bytes - 1) & PAGE_MASK)))
|
||||
return NULL;
|
||||
|
||||
/*
|
||||
* If SNP is enabled, i.e. the RMP is active, allocate a full page to
|
||||
* prevent concurrent accesses to the page. As required by firmware,
|
||||
* the PSP driver updates the RMP to temporarily transfer ownership of
|
||||
* the page to Firmware while the {DE,EN}CRYPT operation is in-progress,
|
||||
* and so concurrent software accesses to the page will encounter
|
||||
* seemingly spurious RMP #PF violations
|
||||
*/
|
||||
if (cc_platform_has(CC_ATTR_HOST_SEV_SNP))
|
||||
return (void *)__get_free_page(GFP_KERNEL);
|
||||
|
||||
return kmalloc(*nr_bytes, GFP_KERNEL);
|
||||
}
|
||||
|
||||
static void sev_dbg_crypt_slow_free(void *buf)
|
||||
{
|
||||
if (cc_platform_has(CC_ATTR_HOST_SEV_SNP))
|
||||
free_page((unsigned long)buf);
|
||||
else
|
||||
kfree(buf);
|
||||
}
|
||||
|
||||
static int sev_dbg_decrypt_slow(struct kvm *kvm, unsigned long src,
|
||||
struct page *src_p, unsigned long dst,
|
||||
unsigned int len, int *err)
|
||||
|
|
@ -1307,7 +1326,7 @@ static int sev_dbg_decrypt_slow(struct kvm *kvm, unsigned long src,
|
|||
if (copy_to_user((void __user *)dst, buf + (src & 15), len))
|
||||
r = -EFAULT;
|
||||
out:
|
||||
kfree(buf);
|
||||
sev_dbg_crypt_slow_free(buf);
|
||||
return r;
|
||||
}
|
||||
|
||||
|
|
@ -1340,7 +1359,7 @@ static int sev_dbg_encrypt_slow(struct kvm *kvm, unsigned long src,
|
|||
r = sev_issue_dbg_cmd(kvm, __sme_set(__pa(buf)), dst_pa,
|
||||
nr_bytes, KVM_SEV_DBG_ENCRYPT, err);
|
||||
out:
|
||||
kfree(buf);
|
||||
sev_dbg_crypt_slow_free(buf);
|
||||
return r;
|
||||
}
|
||||
|
||||
|
|
@ -2757,8 +2776,12 @@ int sev_mem_enc_register_region(struct kvm *kvm,
|
|||
if (!region)
|
||||
return -ENOMEM;
|
||||
|
||||
/*
|
||||
* Do NOT specify FOLL_WRITE, as KVM isn't using the pinned pages to
|
||||
* write memory, and FOLL_LONGTERM itself triggers CoW unshare.
|
||||
*/
|
||||
region->pages = sev_pin_memory(kvm, range->addr, range->size, ®ion->npages,
|
||||
FOLL_WRITE | FOLL_LONGTERM);
|
||||
FOLL_LONGTERM);
|
||||
if (IS_ERR(region->pages)) {
|
||||
ret = PTR_ERR(region->pages);
|
||||
goto e_free;
|
||||
|
|
|
|||
|
|
@ -1179,6 +1179,7 @@ static void init_vmcb(struct kvm_vcpu *vcpu, bool init_event)
|
|||
svm_set_intercept(svm, INTERCEPT_SKINIT);
|
||||
svm_set_intercept(svm, INTERCEPT_WBINVD);
|
||||
svm_set_intercept(svm, INTERCEPT_XSETBV);
|
||||
svm_set_intercept(svm, INTERCEPT_ICEBP);
|
||||
svm_set_intercept(svm, INTERCEPT_RDPRU);
|
||||
svm_set_intercept(svm, INTERCEPT_RSM);
|
||||
|
||||
|
|
@ -1311,11 +1312,6 @@ void svm_switch_vmcb(struct vcpu_svm *svm, struct kvm_vmcb_info *target_vmcb)
|
|||
svm->vmcb = target_vmcb->ptr;
|
||||
}
|
||||
|
||||
static int svm_vcpu_precreate(struct kvm *kvm)
|
||||
{
|
||||
return avic_alloc_physical_id_table(kvm);
|
||||
}
|
||||
|
||||
static int svm_vcpu_create(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
struct vcpu_svm *svm;
|
||||
|
|
@ -2081,6 +2077,22 @@ static int bp_interception(struct kvm_vcpu *vcpu)
|
|||
return 0;
|
||||
}
|
||||
|
||||
static int icebp_interception(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
/*
|
||||
* Intercept and emulate ICEBP (INT1, opcode 0xF1) instead of allowing
|
||||
* the guest to natively take the #DB trap, so that RIP is advanced
|
||||
* past the instruction *before* #DB is injected. This is necessary
|
||||
* because SVM reports the wrong RIP for ICEBP-induced #DB when #DBs
|
||||
* are delivered via a task gate: RIP points at the ICEBP instruction
|
||||
* instead of after it (and SVM doesn't provide enough information for
|
||||
* KVM to detect and manually advance the pre-#DB RIP).
|
||||
*/
|
||||
svm_skip_emulated_instruction(vcpu);
|
||||
kvm_queue_exception(vcpu, DB_VECTOR);
|
||||
return 1;
|
||||
}
|
||||
|
||||
static int ud_interception(struct kvm_vcpu *vcpu)
|
||||
{
|
||||
return handle_ud(vcpu);
|
||||
|
|
@ -3395,6 +3407,7 @@ static int (*const svm_exit_handlers[])(struct kvm_vcpu *vcpu) = {
|
|||
[SVM_EXIT_MONITOR] = kvm_emulate_monitor,
|
||||
[SVM_EXIT_MWAIT] = kvm_emulate_mwait,
|
||||
[SVM_EXIT_XSETBV] = kvm_emulate_xsetbv,
|
||||
[SVM_EXIT_ICEBP] = icebp_interception,
|
||||
[SVM_EXIT_RDPRU] = kvm_handle_invalid_op,
|
||||
[SVM_EXIT_EFER_WRITE_TRAP] = efer_trap,
|
||||
[SVM_EXIT_CR0_WRITE_TRAP] = cr_trap,
|
||||
|
|
@ -5320,12 +5333,6 @@ static int svm_vm_init(struct kvm *kvm)
|
|||
if (!pause_filter_count || !pause_filter_thresh)
|
||||
kvm_disable_exits(kvm, KVM_X86_DISABLE_EXITS_PAUSE);
|
||||
|
||||
if (enable_apicv) {
|
||||
int ret = avic_vm_init(kvm);
|
||||
if (ret)
|
||||
return ret;
|
||||
}
|
||||
|
||||
svm_srso_vm_init();
|
||||
return 0;
|
||||
}
|
||||
|
|
@ -5351,13 +5358,14 @@ struct kvm_x86_ops svm_x86_ops __initdata = {
|
|||
.emergency_disable_virtualization_cpu = svm_emergency_disable_virtualization_cpu,
|
||||
.has_emulated_msr = svm_has_emulated_msr,
|
||||
|
||||
.vcpu_precreate = svm_vcpu_precreate,
|
||||
.vcpu_precreate = avic_vcpu_precreate,
|
||||
.vcpu_create = svm_vcpu_create,
|
||||
.vcpu_free = svm_vcpu_free,
|
||||
.vcpu_reset = svm_vcpu_reset,
|
||||
|
||||
.vm_size = sizeof(struct kvm_svm),
|
||||
.vm_init = svm_vm_init,
|
||||
.vm_pre_destroy = avic_vm_pre_destroy,
|
||||
.vm_destroy = svm_vm_destroy,
|
||||
|
||||
.prepare_switch_to_guest = svm_prepare_switch_to_guest,
|
||||
|
|
@ -5732,6 +5740,8 @@ static __init int svm_hardware_setup(void)
|
|||
enable_apicv = avic_hardware_setup();
|
||||
if (!enable_apicv) {
|
||||
enable_ipiv = false;
|
||||
svm_x86_ops.vcpu_precreate = NULL;
|
||||
svm_x86_ops.vm_pre_destroy = NULL;
|
||||
svm_x86_ops.vcpu_blocking = NULL;
|
||||
svm_x86_ops.vcpu_unblocking = NULL;
|
||||
svm_x86_ops.vcpu_get_apicv_inhibit_reasons = NULL;
|
||||
|
|
|
|||
|
|
@ -947,9 +947,9 @@ extern struct kvm_x86_nested_ops svm_nested_ops;
|
|||
|
||||
bool __init avic_hardware_setup(void);
|
||||
void avic_hardware_unsetup(void);
|
||||
int avic_alloc_physical_id_table(struct kvm *kvm);
|
||||
int avic_vcpu_precreate(struct kvm *kvm);
|
||||
void avic_vm_pre_destroy(struct kvm *kvm);
|
||||
void avic_vm_destroy(struct kvm *kvm);
|
||||
int avic_vm_init(struct kvm *kvm);
|
||||
void avic_init_vmcb(struct vcpu_svm *svm, struct vmcb *vmcb);
|
||||
int avic_incomplete_ipi_interception(struct kvm_vcpu *vcpu);
|
||||
int avic_unaccelerated_access_interception(struct kvm_vcpu *vcpu);
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user