Enable Notify VM exit functionality for TDX guests. Notify VM exit is an existing feature supported by KVM. Userspace can enable Notify VM exit through KVM_CAP_X86_NOTIFY_VMEXIT when it's reported as supported. However, KVM reports the support of this CAP just based on the hardware capability but doesn't differentiate between VMX and TDX. This leads to the issue that userspace can enable this cap for TDX guests without getting an error, but the feature is not actually enabled because KVM doesn't call the TDX module API to program the relevant TD VMCS fields. Enable Notify VM exit for TDX guests by: - Invoking TDX module API calls to set NOTIFY_VM_EXITING and Notify Window in TD VMCS. It's done in tdx_vcpu_init() where other TD VMCS bits are set. Since TDX vCPU cannot be reset, it only needs to be configured once when initializing the TDX vCPU. - Adding corresponding exit handler for TDX Notify VM Exit. Notify VM exit can happen when executing the IRET instruction. If the IRET unblocks the NMI blocking state, bit 12 of the exit qualification is set. In this case, the VMM needs to restore the "blocked by NMI" state when it decides to re-enter the guest. For TDX, KVM cannot manage the GUEST_INTERRUPTIBILITY_INFO and it's TDX module's responsibility to handle it. Extract the common part without NMI blocking handling into a helper in common.h so that it can be shared between VMX and TDX. Note, KVM uses "pre-production" terminology for the feature formally called Notify VM-Exit. All public versions of the SDM refer to the feature as Instruction Timeout. This will be remedied in the near future, for now, use KVM's terminology for consistency. Note, #2, there is no enumeration bit for Notify VM exit by TDX module because all TDX modules support it, and allow to set the corresponding TD VMCS fields as long as the hardware supports the feature. Fixes: 161d34609f9b ("KVM: TDX: Make TDX VM type supported") Cc: stable@vger.kernel.org Signed-off-by: Xiaoyao Li --- Changes in v2: - Mention the feature name mismatch between KVM and SDM in changelog and leave the renaming to future, since this patch is targeted for stable - Extract the common handling into a helper, and put the helper in common.h instead of refactorin the existing handle_notify() in vmx.h - Add a note to clarify the feature is always supported by TDX module, to make Sashiko happy. --- arch/x86/kvm/vmx/common.h | 19 +++++++++++++++++++ arch/x86/kvm/vmx/tdx.c | 10 ++++++++++ arch/x86/kvm/vmx/vmx.c | 13 +------------ 3 files changed, 30 insertions(+), 12 deletions(-) diff --git a/arch/x86/kvm/vmx/common.h b/arch/x86/kvm/vmx/common.h index 08005676702c..2cbaa9aba901 100644 --- a/arch/x86/kvm/vmx/common.h +++ b/arch/x86/kvm/vmx/common.h @@ -4,6 +4,7 @@ #include #include +#include #include "mmu.h" @@ -183,6 +184,24 @@ static inline void __vmx_deliver_posted_interrupt(struct kvm_vcpu *vcpu, kvm_vcpu_trigger_posted_interrupt(vcpu, POSTED_INTR_VECTOR); } +static inline int __vmx_handle_notify(struct kvm_vcpu *vcpu, + unsigned long exit_qual) +{ + bool context_invalid = exit_qual & NOTIFY_VM_CONTEXT_INVALID; + + ++vcpu->stat.notify_window_exits; + + if (vcpu->kvm->arch.notify_vmexit_flags & KVM_X86_NOTIFY_VMEXIT_USER || + context_invalid) { + vcpu->run->exit_reason = KVM_EXIT_NOTIFY; + vcpu->run->notify.flags = context_invalid ? + KVM_NOTIFY_CONTEXT_INVALID : 0; + return 0; + } + + return 1; +} + noinstr void vmx_handle_nmi(struct kvm_vcpu *vcpu); #endif /* __KVM_X86_VMX_COMMON_H */ diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index b272c20586a7..7338ac0af693 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -2126,6 +2126,9 @@ int tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) * - If it's not an MSMI, no need to do anything here. */ return 1; + case EXIT_REASON_NOTIFY: + /* NMI blocking state is handled by TDX module */ + return __vmx_handle_notify(vcpu, vmx_get_exit_qual(vcpu)); default: break; } @@ -3154,6 +3157,13 @@ static int tdx_vcpu_init(struct kvm_vcpu *vcpu, struct kvm_tdx_cmd *cmd) td_vmcs_write64(tdx, POSTED_INTR_DESC_ADDR, __pa(&tdx->vt.pi_desc)); td_vmcs_setbit32(tdx, PIN_BASED_VM_EXEC_CONTROL, PIN_BASED_POSTED_INTR); + if (kvm_notify_vmexit_enabled(vcpu->kvm)) { + td_vmcs_setbit32(tdx, SECONDARY_VM_EXEC_CONTROL, + SECONDARY_EXEC_NOTIFY_VM_EXITING); + td_vmcs_write32(tdx, NOTIFY_WINDOW, + vcpu->kvm->arch.notify_window); + } + tdx->state = VCPU_TD_STATE_INITIALIZED; return 0; diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c index e3bfe6aca1a0..e53cc96002c7 100644 --- a/arch/x86/kvm/vmx/vmx.c +++ b/arch/x86/kvm/vmx/vmx.c @@ -6279,9 +6279,6 @@ static int handle_bus_lock_vmexit(struct kvm_vcpu *vcpu) static int handle_notify(struct kvm_vcpu *vcpu) { unsigned long exit_qual = vmx_get_exit_qual(vcpu); - bool context_invalid = exit_qual & NOTIFY_VM_CONTEXT_INVALID; - - ++vcpu->stat.notify_window_exits; /* * Notify VM exit happened while executing iret from NMI, @@ -6291,15 +6288,7 @@ static int handle_notify(struct kvm_vcpu *vcpu) vmcs_set_bits(GUEST_INTERRUPTIBILITY_INFO, GUEST_INTR_STATE_NMI); - if (vcpu->kvm->arch.notify_vmexit_flags & KVM_X86_NOTIFY_VMEXIT_USER || - context_invalid) { - vcpu->run->exit_reason = KVM_EXIT_NOTIFY; - vcpu->run->notify.flags = context_invalid ? - KVM_NOTIFY_CONTEXT_INVALID : 0; - return 0; - } - - return 1; + return __vmx_handle_notify(vcpu, exit_qual); } static int vmx_get_msr_imm_reg(struct kvm_vcpu *vcpu) -- 2.43.0 Get and check the exit reason from the low 16 bits of vp_enter_ret, and store the synthesized/transformed exit reason in the "basic" field. Some bits in the upper 16 bits in the exit reason have their own meanings and they might be 1. When handling the exit reason, only do handling on the lower 16 bits and keep the upper 16 bits unchanged. This change also helps remove the additional check in tdx_failed_vmentry(). Note, due to the synthesized invalid exit reason, -1, is changed to assigned to the "basic" field, adjust the checking in tdx_get_exit_info() accordingly. Fixes: 095b71a03f49 ("KVM: TDX: Add a place holder to handle TDX VM exit") Fixes: c42856af8f70 ("KVM: TDX: Add a place holder for handler of TDX hypercalls (TDG.VP.VMCALL)") Cc: stable@vger.kernel.org Signed-off-by: Xiaoyao Li --- Changes in v2: - new patch --- arch/x86/kvm/vmx/tdx.c | 35 ++++++++++++++++++++++------------- 1 file changed, 22 insertions(+), 13 deletions(-) diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index 7338ac0af693..a89885d550c9 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -921,10 +921,10 @@ static __always_inline u32 tdcall_to_vmx_exit_reason(struct kvm_vcpu *vcpu) return EXIT_REASON_TDCALL; } -static __always_inline u32 tdx_to_vmx_exit_reason(struct kvm_vcpu *vcpu) +static __always_inline union vmx_exit_reason tdx_to_vmx_exit_reason(struct kvm_vcpu *vcpu) { struct vcpu_tdx *tdx = to_tdx(vcpu); - u32 exit_reason; + union vmx_exit_reason exit_reason; switch (tdx->vp_enter_ret & TDX_SEAMCALL_STATUS_MASK) { case TDX_SUCCESS: @@ -934,23 +934,33 @@ static __always_inline u32 tdx_to_vmx_exit_reason(struct kvm_vcpu *vcpu) case TDX_NON_RECOVERABLE_TD_WRONG_APIC_MODE: break; default: - return -1u; + /* + * Synthesize an invalid bogus Exit Reason, as the TDX-Module + * never attempted to run the vCPU, i.e. the Exit Reason is + * undefined, but this is NOT a failed VM-Enter. + */ + return (union vmx_exit_reason) { + .basic = -1, + }; } - exit_reason = tdx->vp_enter_ret; + exit_reason.full = (u32)tdx->vp_enter_ret; - switch (exit_reason) { + switch (exit_reason.basic) { case EXIT_REASON_TDCALL: if (tdvmcall_exit_type(vcpu)) - return EXIT_REASON_VMCALL; - - return tdcall_to_vmx_exit_reason(vcpu); + exit_reason.basic = EXIT_REASON_VMCALL; + else + exit_reason.basic = tdcall_to_vmx_exit_reason(vcpu); + break; case EXIT_REASON_EPT_MISCONFIG: /* * Defer KVM_BUG_ON() until tdx_handle_exit() because this is in * non-instrumentable code with interrupts disabled. */ - return -1u; + return (union vmx_exit_reason) { + .basic = -1, + }; default: break; } @@ -967,7 +977,7 @@ static noinstr void tdx_vcpu_enter_exit(struct kvm_vcpu *vcpu) tdx->vp_enter_ret = tdh_vp_enter(&tdx->vp, &tdx->vp_enter_args); - vt->exit_reason.full = tdx_to_vmx_exit_reason(vcpu); + vt->exit_reason = tdx_to_vmx_exit_reason(vcpu); vt->exit_qualification = tdx->vp_enter_args.rcx; tdx->ext_exit_qualification = tdx->vp_enter_args.rdx; @@ -981,8 +991,7 @@ static noinstr void tdx_vcpu_enter_exit(struct kvm_vcpu *vcpu) static bool tdx_failed_vmentry(struct kvm_vcpu *vcpu) { - return vmx_get_exit_reason(vcpu).failed_vmentry && - vmx_get_exit_reason(vcpu).full != -1u; + return vmx_get_exit_reason(vcpu).failed_vmentry; } static fastpath_t tdx_exit_handlers_fastpath(struct kvm_vcpu *vcpu) @@ -2144,7 +2153,7 @@ void tdx_get_exit_info(struct kvm_vcpu *vcpu, u32 *reason, struct vcpu_tdx *tdx = to_tdx(vcpu); *reason = tdx->vt.exit_reason.full; - if (*reason != -1u) { + if (tdx->vt.exit_reason.basic != -1) { *info1 = vmx_get_exit_qual(vcpu); *info2 = tdx->ext_exit_qualification; *intr_info = vmx_get_intr_info(vcpu); -- 2.43.0 Enable Bus Lock VM exit functionality for TDX guests. Userspace can enable KVM_BUS_LOCK_DETECTION_EXIT for TDX guests without getting an error, but the feature is not actually enabled because KVM does not yet program the TDX execution control or handle the resulting exit. Enable Bus Lock VM exit for TDX guests by programming the BUS_LOCK_DETECTION control in the TD VMCS and by adding the exit handler. Clear the bus_lock_detected bit to avoid being counted multiple times if it needs to return early for wait_for_sept_zap case in tdx_vcpu_run(). Since the wait_for_sept_zap case is expected to be rare, just do the clearing of bus_lock_detected unconditionally. Note, there is no enumeration bit for this feature by TDX module because all TDX modules support it, and allow to set the TD VMCS as long as the hardware supports the feature. Fixes: 161d34609f9b ("KVM: TDX: Make TDX VM type supported") Cc: stable@vger.kernel.org Originally-by: Chenyi Qiang Signed-off-by: Xiaoyao Li --- Changes in v2: - Don't overwrite the negative return value to 0. (Sashiko) - Clear the bus_lock_detected bit when it returns early for wait_for_sept_zap case. - Add a note to clarify the feature is always supported by the TDX module, to make Sashiko happy. --- arch/x86/kvm/vmx/tdx.c | 28 ++++++++++++++++++++++++++-- arch/x86/kvm/vmx/vmx.c | 2 +- arch/x86/kvm/vmx/vmx.h | 1 + 3 files changed, 28 insertions(+), 3 deletions(-) diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index a89885d550c9..ac3f71643cd5 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -1080,8 +1080,10 @@ fastpath_t tdx_vcpu_run(struct kvm_vcpu *vcpu, u64 run_flags) * allowing vCPU entry to avoid contention with tdh_vp_enter() and * TDCALLs. */ - if (unlikely(READ_ONCE(to_kvm_tdx(vcpu->kvm)->wait_for_sept_zap))) + if (unlikely(READ_ONCE(to_kvm_tdx(vcpu->kvm)->wait_for_sept_zap))) { + vt->exit_reason.bus_lock_detected = 0; return EXIT_FASTPATH_EXIT_HANDLED; + } trace_kvm_entry(vcpu, run_flags & KVM_RUN_FORCE_IMMEDIATE_EXIT); @@ -2037,7 +2039,7 @@ int tdx_complete_emulated_msr(struct kvm_vcpu *vcpu, int err) } -int tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) +static int __tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) { struct vcpu_tdx *tdx = to_tdx(vcpu); u64 vp_enter_ret = tdx->vp_enter_ret; @@ -2138,6 +2140,8 @@ int tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) case EXIT_REASON_NOTIFY: /* NMI blocking state is handled by TDX module */ return __vmx_handle_notify(vcpu, vmx_get_exit_qual(vcpu)); + case EXIT_REASON_BUS_LOCK: + return handle_bus_lock_vmexit(vcpu); default: break; } @@ -2147,6 +2151,22 @@ int tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) return 0; } +int tdx_handle_exit(struct kvm_vcpu *vcpu, fastpath_t fastpath) +{ + int ret = __tdx_handle_exit(vcpu, fastpath); + + /* Exit to user space when bus lock was detected */ + if (vmx_get_exit_reason(vcpu).bus_lock_detected) { + if (ret > 0) { + vcpu->run->exit_reason = KVM_EXIT_X86_BUS_LOCK; + ret = 0; + } + + vcpu->run->flags |= KVM_RUN_X86_BUS_LOCK; + } + return ret; +} + void tdx_get_exit_info(struct kvm_vcpu *vcpu, u32 *reason, u64 *info1, u64 *info2, u32 *intr_info, u32 *error_code) { @@ -3173,6 +3193,10 @@ static int tdx_vcpu_init(struct kvm_vcpu *vcpu, struct kvm_tdx_cmd *cmd) vcpu->kvm->arch.notify_window); } + if (vcpu->kvm->arch.bus_lock_detection_enabled) + td_vmcs_setbit32(tdx, SECONDARY_VM_EXEC_CONTROL, + SECONDARY_EXEC_BUS_LOCK_DETECTION); + tdx->state = VCPU_TD_STATE_INITIALIZED; return 0; diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c index e53cc96002c7..c429db9b9205 100644 --- a/arch/x86/kvm/vmx/vmx.c +++ b/arch/x86/kvm/vmx/vmx.c @@ -6265,7 +6265,7 @@ static int handle_encls(struct kvm_vcpu *vcpu) } #endif /* CONFIG_X86_SGX_KVM */ -static int handle_bus_lock_vmexit(struct kvm_vcpu *vcpu) +int handle_bus_lock_vmexit(struct kvm_vcpu *vcpu) { /* * Hardware may or may not set the BUS_LOCK_DETECTED flag on BUS_LOCK diff --git a/arch/x86/kvm/vmx/vmx.h b/arch/x86/kvm/vmx/vmx.h index dc8517f15bc4..8faf04c09721 100644 --- a/arch/x86/kvm/vmx/vmx.h +++ b/arch/x86/kvm/vmx/vmx.h @@ -379,6 +379,7 @@ bool __vmx_vcpu_run(struct vcpu_vmx *vmx, unsigned int flags); void vmx_ept_load_pdptrs(struct kvm_vcpu *vcpu); void vmx_set_intercept_for_msr(struct kvm_vcpu *vcpu, u32 msr, int type, bool set); +int handle_bus_lock_vmexit(struct kvm_vcpu *vcpu); static inline void vmx_disable_intercept_for_msr(struct kvm_vcpu *vcpu, u32 msr, int type) -- 2.43.0