KVM VMX changes for 7.3

- Service local TLB flushes on a failed nested VM-Enter to fix a bug where KVM
    could miss a TLB on a future, successful VM-Enter with the same L2 VPID.
 
  - Cap the maximum value shoved into the VMX Preemption Timer to workaround an
    erratum that affects all existing Intel CPUs that support CPUID 0x15.
 -----BEGIN PGP SIGNATURE-----
 
 iQIzBAABCgAdFiEEKTobbabEP7vbhhN9OlYIJqCjN/0FAmp82CwACgkQOlYIJqCj
 N/1ekg//Z3ulUOnxKhywPtJ9T2bGlGoqJwslTygFy2JQfR4eNdnNYvtLU2A6Gsa2
 1IaB/xJqMtSb+XxpgghYuGvU29J/i25QOJRM0RyJhaXxYvDkZPP4P1gwm9/qniFn
 OTITTM37twD98uIYMjOXP6UsK7MPg2Kogv74hYwTpHVj/ti9weLwRX+x+bmlA2Aj
 lPFUqPqQkLWFvhoSLV5uYW1rZfLtq/UdlV13DCF+vtp1u6IXaW6mFG5RMFog9kSt
 arIW2ZG9Gx7SwrLrpRUSpQyRZugzTcNyl/lK+/sGzONfJO1Gw2QZ7bJFckhXoRt1
 ktR5ZAXc8Zy7It7gsNozvQD1NsnT2+FTDzygv7h15IIOwWi+quWF63BCNz+in5RA
 ivmFrUnRwmbGgoshpD+JjwR5qW7TK0lOLudWMWnH6ctIH3PvgJrNiIuTbKNGxtkI
 D0WuDLrLcSqiR9M8q5eLTKiRjykYugG81KIHIMoTBszpO8/qIhdXyPUr3TPUkoHR
 SeDL0m4KdURZR2cOTOvy6DIPTY2qn+rpxRFkqXJCUx/98kajsF0/0RDMBwUeaGF8
 Y9e9bgR7YW8hMd6z64ODFTLUSMYXavC0AJpQq3spWg0bo15CcAFTAAnyAQOpTgyo
 IGENMWREkCa62RrveZLhN3rFkYBbGMZSuBa0UQM+NvvJXDwdJ90=
 =BvM6
 -----END PGP SIGNATURE-----

Merge tag 'kvm-x86-vmx-7.3' of https://github.com/kvm-x86/linux into HEAD

KVM VMX changes for 7.3

 - Service local TLB flushes on a failed nested VM-Enter to fix a bug where KVM
   could miss a TLB on a future, successful VM-Enter with the same L2 VPID.

 - Cap the maximum value shoved into the VMX Preemption Timer to workaround an
   erratum that affects all existing Intel CPUs that support CPUID 0x15.
This commit is contained in:
Paolo Bonzini 2026-08-18 13:30:24 +02:00
commit 634c347df1
2 changed files with 117 additions and 62 deletions

View File

@ -3759,6 +3759,14 @@ enum nvmx_vmentry_status nested_vmx_enter_non_root_mode(struct kvm_vcpu *vcpu,
vmentry_fail_vmexit_guest_mode:
if (vmcs12->cpu_based_vm_exec_control & CPU_BASED_USE_TSC_OFFSETTING)
vcpu->arch.tsc_offset -= vmcs12->tsc_offset;
/*
* Handle any TLB flush requests that were queued for L2 if KVM made it
* far enough along to switch to L2 context. Note, loading host state
* will generate any flushes for L1 required by VM-Exit.
*/
kvm_service_local_tlb_flush_requests(vcpu);
leave_guest_mode(vcpu);
vmentry_fail_vmexit:

View File

@ -150,10 +150,13 @@ module_param(dump_invalid_vmcs, bool, 0644);
#define KVM_VMX_TSC_MULTIPLIER_MAX 0xffffffffffffffffULL
/* Guest_tsc -> host_tsc conversion requires 64-bit division. */
#ifdef CONFIG_X86_64
static int __read_mostly cpu_preemption_timer_multi;
static bool __read_mostly enable_preemption_timer = 1;
#ifdef CONFIG_X86_64
static u64 __ro_after_init preemption_timer_max_value;
module_param_named(preemption_timer, enable_preemption_timer, bool, S_IRUGO);
#else
#define enable_preemption_timer false
#endif
extern bool __read_mostly allow_smaller_maxphyaddr;
@ -2159,7 +2162,7 @@ int vmx_get_msr(struct kvm_vcpu *vcpu, struct msr_data *msr_info)
!guest_has_spec_ctrl_msr(vcpu))
return 1;
msr_info->data = to_vmx(vcpu)->spec_ctrl;
msr_info->data = vmx->spec_ctrl;
break;
case MSR_IA32_SYSENTER_CS:
msr_info->data = vmcs_read32(GUEST_SYSENTER_CS);
@ -2191,7 +2194,7 @@ int vmx_get_msr(struct kvm_vcpu *vcpu, struct msr_data *msr_info)
if (!msr_info->host_initiated &&
!guest_cpu_cap_has(vcpu, X86_FEATURE_SGX_LC))
return 1;
msr_info->data = to_vmx(vcpu)->msr_ia32_sgxlepubkeyhash
msr_info->data = vmx->msr_ia32_sgxlepubkeyhash
[msr_info->index - MSR_IA32_SGXLEPUBKEYHASH0];
break;
case KVM_FIRST_EMULATED_VMX_MSR ... KVM_LAST_EMULATED_VMX_MSR:
@ -2404,7 +2407,7 @@ int vmx_set_msr(struct kvm_vcpu *vcpu, struct msr_data *msr_info)
vmx_guest_debugctl_write(vcpu, data);
if (intel_pmu_lbr_is_enabled(vcpu) && !to_vmx(vcpu)->lbr_desc.event &&
if (intel_pmu_lbr_is_enabled(vcpu) && !vmx->lbr_desc.event &&
(data & DEBUGCTLMSR_LBR))
intel_pmu_create_guest_lbr_event(vcpu);
return 0;
@ -2483,7 +2486,7 @@ int vmx_set_msr(struct kvm_vcpu *vcpu, struct msr_data *msr_info)
break;
case MSR_IA32_MCG_EXT_CTL:
if ((!msr_info->host_initiated &&
!(to_vmx(vcpu)->msr_ia32_feature_control &
!(vmx->msr_ia32_feature_control &
FEAT_CTL_LMCE_ENABLED)) ||
(data & ~MCG_EXT_CTL_LMCE_EN))
return 1;
@ -3678,13 +3681,14 @@ void vmx_get_segment(struct kvm_vcpu *vcpu, struct kvm_segment *var, int seg)
u64 vmx_get_segment_base(struct kvm_vcpu *vcpu, int seg)
{
struct vcpu_vmx *vmx = to_vmx(vcpu);
struct kvm_segment s;
if (to_vmx(vcpu)->rmode.vm86_active) {
if (vmx->rmode.vm86_active) {
vmx_get_segment(vcpu, &s, seg);
return s.base;
}
return vmx_read_guest_seg_base(to_vmx(vcpu), seg);
return vmx_read_guest_seg_base(vmx, seg);
}
static int __vmx_get_cpl(struct kvm_vcpu *vcpu, bool no_cache)
@ -7407,32 +7411,6 @@ static void vmx_refresh_guest_perf_global_control(struct kvm_vcpu *vcpu)
pmu->global_ctrl = vmcs_read64(GUEST_IA32_PERF_GLOBAL_CTRL);
}
static void vmx_update_hv_timer(struct kvm_vcpu *vcpu, bool force_immediate_exit)
{
struct vcpu_vmx *vmx = to_vmx(vcpu);
u64 tscl;
u32 delta_tsc;
if (force_immediate_exit) {
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, 0);
vmx->loaded_vmcs->hv_timer_soft_disabled = false;
} else if (vmx->hv_deadline_tsc != -1) {
tscl = rdtsc();
if (vmx->hv_deadline_tsc > tscl)
/* set_hv_timer ensures the delta fits in 32-bits */
delta_tsc = (u32)((vmx->hv_deadline_tsc - tscl) >>
cpu_preemption_timer_multi);
else
delta_tsc = 0;
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, delta_tsc);
vmx->loaded_vmcs->hv_timer_soft_disabled = false;
} else if (!vmx->loaded_vmcs->hv_timer_soft_disabled) {
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, -1);
vmx->loaded_vmcs->hv_timer_soft_disabled = true;
}
}
void noinstr vmx_update_host_rsp(struct vcpu_vmx *vmx, unsigned long host_rsp)
{
if (unlikely(host_rsp != vmx->loaded_vmcs->host_state.rsp)) {
@ -7518,6 +7496,8 @@ static noinstr void vmx_vcpu_enter_exit(struct kvm_vcpu *vcpu,
guest_state_exit_irqoff();
}
static void vmx_update_hv_timer(struct kvm_vcpu *vcpu, bool force_immediate_exit);
fastpath_t vmx_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags)
{
bool force_immediate_exit = run_flags & KVM_RUN_FORCE_IMMEDIATE_EXIT;
@ -8330,6 +8310,65 @@ static inline int u64_shl_div_u64(u64 a, unsigned int shift,
return 0;
}
/*
* Workaround for a widespread Intel erratum (e.g. EMR158) where the
* VMX-preemption timer may expire earlier than expected when programmed
* with large values. The workaround is to cap the timer value to strictly
* less than 2^25 * CPUID.15H:EBX / CPUID.15H:EAX.
*/
static __init u64 calc_preemption_timer_max_value(void)
{
const u64 ARCHITECTURAL_MAX_VALUE = UINT_MAX;
u32 eax, ebx, ecx, edx;
if (cpu_feature_enabled(X86_FEATURE_HYPERVISOR))
return ARCHITECTURAL_MAX_VALUE;
if (cpuid_eax(0) < 0x15)
return ARCHITECTURAL_MAX_VALUE;
cpuid(0x15, &eax, &ebx, &ecx, &edx);
if (!eax || !ebx)
return ARCHITECTURAL_MAX_VALUE;
if (WARN_ON_ONCE(!(((u64)ebx << 25) / eax)))
return ARCHITECTURAL_MAX_VALUE;
return min((((u64)ebx << 25) / eax) - 1, ARCHITECTURAL_MAX_VALUE);
}
static __init void vmx_setup_preemption_timer(void)
{
if (!cpu_has_vmx_preemption_timer())
enable_preemption_timer = false;
if (enable_preemption_timer) {
u64 use_timer_freq = 5000ULL * 1000 * 1000;
cpu_preemption_timer_multi =
vmx_misc_preemption_timer_rate(vmcs_config.misc);
preemption_timer_max_value = calc_preemption_timer_max_value();
if (tsc_khz)
use_timer_freq = (u64)tsc_khz * 1000;
use_timer_freq >>= cpu_preemption_timer_multi;
/*
* KVM "disables" the preemption timer by setting it to its max
* value. Don't use the timer if it might cause spurious exits
* at a rate faster than 0.1 Hz (of uninterrupted guest time).
*/
if (use_timer_freq > preemption_timer_max_value / 10)
enable_preemption_timer = false;
}
if (!enable_preemption_timer) {
vt_x86_ops.set_hv_timer = NULL;
vt_x86_ops.cancel_hv_timer = NULL;
}
}
int vmx_set_hv_timer(struct kvm_vcpu *vcpu, u64 guest_deadline_tsc,
bool *expired)
{
@ -8357,12 +8396,12 @@ int vmx_set_hv_timer(struct kvm_vcpu *vcpu, u64 guest_deadline_tsc,
return -ERANGE;
/*
* If the delta tsc can't fit in the 32 bit after the multi shift,
* we can't use the preemption timer.
* If the delta tsc exceeds the preemption timer limit after the
* multi shift, we can't use the preemption timer.
* It's possible that it fits on later vmentries, but checking
* on every vmentry is costly so we just use an hrtimer.
*/
if (delta_tsc >> (cpu_preemption_timer_multi + 32))
if ((delta_tsc >> cpu_preemption_timer_multi) > preemption_timer_max_value)
return -ERANGE;
vmx->hv_deadline_tsc = tscl + delta_tsc;
@ -8374,6 +8413,39 @@ void vmx_cancel_hv_timer(struct kvm_vcpu *vcpu)
{
to_vmx(vcpu)->hv_deadline_tsc = -1;
}
static void vmx_update_hv_timer(struct kvm_vcpu *vcpu, bool force_immediate_exit)
{
struct vcpu_vmx *vmx = to_vmx(vcpu);
u64 tscl;
u32 delta_tsc;
if (force_immediate_exit) {
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, 0);
vmx->loaded_vmcs->hv_timer_soft_disabled = false;
} else if (vmx->hv_deadline_tsc != -1) {
tscl = rdtsc();
if (vmx->hv_deadline_tsc > tscl)
/* set_hv_timer ensures the delta fits in 32-bits */
delta_tsc = (u32)((vmx->hv_deadline_tsc - tscl) >>
cpu_preemption_timer_multi);
else
delta_tsc = 0;
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, delta_tsc);
vmx->loaded_vmcs->hv_timer_soft_disabled = false;
} else if (!vmx->loaded_vmcs->hv_timer_soft_disabled) {
vmcs_write32(VMX_PREEMPTION_TIMER_VALUE, preemption_timer_max_value);
vmx->loaded_vmcs->hv_timer_soft_disabled = true;
}
}
#else
static __init void vmx_setup_preemption_timer(void) { }
static void vmx_update_hv_timer(struct kvm_vcpu *vcpu, bool force_immediate_exit)
{
BUILD_BUG_ON(1);
}
#endif
void vmx_update_cpu_dirty_logging(struct kvm_vcpu *vcpu)
@ -8736,32 +8808,7 @@ __init int vmx_hardware_setup(void)
if (!enable_ept || !enable_ept_ad_bits || !cpu_has_vmx_pml())
enable_pml = 0;
if (!cpu_has_vmx_preemption_timer())
enable_preemption_timer = false;
if (enable_preemption_timer) {
u64 use_timer_freq = 5000ULL * 1000 * 1000;
cpu_preemption_timer_multi =
vmx_misc_preemption_timer_rate(vmcs_config.misc);
if (tsc_khz)
use_timer_freq = (u64)tsc_khz * 1000;
use_timer_freq >>= cpu_preemption_timer_multi;
/*
* KVM "disables" the preemption timer by setting it to its max
* value. Don't use the timer if it might cause spurious exits
* at a rate faster than 0.1 Hz (of uninterrupted guest time).
*/
if (use_timer_freq > 0xffffffffu / 10)
enable_preemption_timer = false;
}
if (!enable_preemption_timer) {
vt_x86_ops.set_hv_timer = NULL;
vt_x86_ops.cancel_hv_timer = NULL;
}
vmx_setup_preemption_timer();
kvm_caps.supported_mce_cap |= MCG_LMCE_P;
kvm_caps.supported_mce_cap |= MCG_CMCI_P;