From 4ae2401e3d6e33b813abdb99469d6f3da2019e1d Mon Sep 17 00:00:00 2001 From: tonic Date: Wed, 23 Sep 2026 22:43:45 +0800 Subject: [PATCH] hypervisor: kvm: save and restore the guest shadow stack pointer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On a host whose KVM exposes CET shadow stacks to the guest (CPUID.(EAX=7,ECX=0):ECX[7]; e.g. Linux 7.0 on AMD Zen 3), a Windows guest bugchecks 0x50 PAGE_FAULT_IN_NONPAGED_AREA at 0xfffffffffffffff8, from nt!KePopulateContinuationContext, right after being restored from some snapshots — about one in five on our node. A vCPU paused while running a user thread with shadow stacks enabled keeps its live SSP in the SSP register, not in MSR_IA32_PL3_SSP. KVM exposes that register only as KVM_REG_GUEST_SSP through KVM_{GET,SET}_ONE_REG, which Cloud Hypervisor never saved. The thread therefore resumed with SSP=0, took a shadow-stack #PF on its next CALL/RET, and the kernel faulted again reading 0 - 8 while building the exception's continuation context. Snapshots taken while every vCPU was in the kernel or running a thread without shadow stacks were unaffected, which is why only some of them failed. Save the register in VcpuKvmState when the guest CPUID exposes shadow stacks and restore it after the MSRs, as QEMU does. The field is optional, so snapshots taken before this change still restore, with the old behaviour. Signed-off-by: tonic --- hypervisor/src/kvm/mod.rs | 68 +++++++++++++++++++++++++++++++- hypervisor/src/kvm/x86_64/mod.rs | 11 ++++++ 2 files changed, 77 insertions(+), 2 deletions(-) diff --git a/hypervisor/src/kvm/mod.rs b/hypervisor/src/kvm/mod.rs index f48a620d5f..f7d0cb4338 100644 --- a/hypervisor/src/kvm/mod.rs +++ b/hypervisor/src/kvm/mod.rs @@ -82,9 +82,9 @@ use kvm_bindings::{ kvm_msr_entry, }; #[cfg(target_arch = "x86_64")] -use x86_64::check_required_kvm_extensions; -#[cfg(target_arch = "x86_64")] pub use x86_64::{CpuId, ExtendedControlRegisters, MsrEntries, VcpuKvmState}; +#[cfg(target_arch = "x86_64")] +use x86_64::{check_required_kvm_extensions, cpuid_has_shstk}; #[cfg(target_arch = "x86_64")] use crate::ClockData; @@ -202,6 +202,25 @@ ioctl_iow_nr!( 0xe3, kvm_bindings::kvm_device_attr ); +// kvm-ioctls only exposes KVM_{GET,SET}_ONE_REG for aarch64 and riscv64. +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_GET_ONE_REG, + kvm_bindings::KVMIO, + 0xab, + kvm_bindings::kvm_one_reg +); +#[cfg(target_arch = "x86_64")] +ioctl_iow_nr!( + KVM_SET_ONE_REG, + kvm_bindings::KVMIO, + 0xac, + kvm_bindings::kvm_one_reg +); +// KVM_X86_REG_KVM(KVM_REG_GUEST_SSP): the guest's current shadow stack pointer. It is +// a register, not an MSR, so KVM_GET_MSRS never carries it. +#[cfg(target_arch = "x86_64")] +const KVM_REG_GUEST_SSP: u64 = 0x2030_0003_0000_0000; #[cfg(feature = "sev_snp")] use igvm_defs::PAGE_SIZE_4K; @@ -2077,6 +2096,45 @@ impl KvmVcpu { let ret = unsafe { ioctl_with_ref(&self.fd, KVM_HAS_DEVICE_ATTR(), &attr) }; ret == 0 } + + /// The guest's current shadow stack pointer, or `None` when the guest CPUID + /// does not expose shadow stacks. A vCPU paused in user mode keeps its live SSP + /// here, not in MSR_IA32_PL3_SSP, so without it a restored thread resumes with + /// SSP=0 and faults on its next CALL/RET. + fn guest_ssp(&self, cpuid: &[CpuIdEntry]) -> cpu::Result> { + if !cpuid_has_shstk(cpuid) { + return Ok(None); + } + let mut ssp = 0u64; + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw mut ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, filled in by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_GET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::GetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(Some(ssp)) + } + + /// Restore the guest's current shadow stack pointer. + fn set_guest_ssp(&self, ssp: u64) -> cpu::Result<()> { + let reg = kvm_bindings::kvm_one_reg { + id: KVM_REG_GUEST_SSP, + addr: &raw const ssp as u64, + }; + // SAFETY: FFI call; `reg.addr` points to `ssp`, read by the kernel. + let ret = unsafe { ioctl_with_ref(&self.fd, KVM_SET_ONE_REG(), ®) }; + if ret < 0 { + return Err(cpu::HypervisorCpuError::SetRegister( + io::Error::last_os_error().into(), + )); + } + Ok(()) + } } /// Implementation of Vcpu trait for KVM @@ -3243,6 +3301,7 @@ impl cpu::Vcpu for KvmVcpu { let vcpu_events = self.get_vcpu_events()?; let tsc_khz = self.tsc_khz()?; + let guest_ssp = self.guest_ssp(&cpuid)?; Ok(VcpuKvmState { cpuid, @@ -3258,6 +3317,7 @@ impl cpu::Vcpu for KvmVcpu { tsc_khz, nested_state, hyperv_synic, + guest_ssp, } .into()) } @@ -3497,6 +3557,10 @@ impl cpu::Vcpu for KvmVcpu { } } + if let Some(ssp) = state.guest_ssp { + self.set_guest_ssp(ssp)?; + } + self.set_vcpu_events(&state.vcpu_events)?; Ok(()) diff --git a/hypervisor/src/kvm/x86_64/mod.rs b/hypervisor/src/kvm/x86_64/mod.rs index 5bc3ee89a0..c0a49ff310 100644 --- a/hypervisor/src/kvm/x86_64/mod.rs +++ b/hypervisor/src/kvm/x86_64/mod.rs @@ -87,6 +87,17 @@ pub struct VcpuKvmState { pub nested_state: Option, #[serde(default)] pub hyperv_synic: bool, + // Absent from snapshots taken before it was saved, and when the guest has no + // shadow stacks. + #[serde(default)] + pub guest_ssp: Option, +} + +/// CPUID.(EAX=7,ECX=0):ECX[7], CET shadow stack. +pub fn cpuid_has_shstk(cpuid: &[CpuIdEntry]) -> bool { + cpuid + .iter() + .any(|e| e.function == 7 && e.index == 0 && e.ecx & (1 << 7) != 0) } impl From for kvm_segment {