[RFC PATCH 35/46] KVM: x86: Implement Caretaker VM-exit dispatch and instruction decoders

Pasha Tatashin pasha.tatashin at soleen.com
Sun Sep 20 12:36:39 PDT 2026


Implement standalone Caretaker VM-exit decoding, MSR/CPUID handling,
and I/O port emulation in arch/x86/kvm/caretaker.c.

Signed-off-by: Pasha Tatashin <pasha.tatashin at soleen.com>
---
 arch/x86/kvm/caretaker.c | 593 +++++++++++++++++++++++++++++++++++++++
 1 file changed, 593 insertions(+)

diff --git a/arch/x86/kvm/caretaker.c b/arch/x86/kvm/caretaker.c
index 9e0f20cbb137..dd2b2d582b69 100644
--- a/arch/x86/kvm/caretaker.c
+++ b/arch/x86/kvm/caretaker.c
@@ -260,6 +260,599 @@ static void kvm_x86_caretaker_init_gdt_tss(struct desc_struct *gdt,
 	caretaker_set_tss_desc(gdt, (unsigned long)tss, sizeof(struct x86_hw_tss) - 1);
 }
 
+static void __cpu_preserved_text
+kvm_x86_caretaker_load_desc(struct desc_struct *gdt, size_t gdt_size,
+			    gate_desc *idt, size_t idt_size,
+			    void *tss)
+{
+	struct desc_ptr gdt_desc = {
+		.size = gdt_size - 1,
+		.address = (unsigned long)gdt,
+	};
+	struct desc_ptr idt_desc = {
+		.size = idt_size - 1,
+		.address = (unsigned long)idt,
+	};
+
+	caretaker_set_tss_desc(gdt, (unsigned long)tss, sizeof(struct x86_hw_tss) - 1);
+	load_gdt(&gdt_desc);
+	native_load_idt(&idt_desc);
+	asm volatile("ltr %w0" : : "q" ((u16)(GDT_ENTRY_TSS * 8)));
+}
+
+static __caretaker_text void
+kvm_x86_caretaker_save_host_state(struct caretaker_x86_host_state *host,
+				  struct caretaker_x86_page *cxp)
+{
+	struct cpu_preserved_stack_context *sctx;
+	u64 apic_base;
+
+	store_idt(&host->orig_idt);
+	host->orig_cr2 = native_read_cr2();
+	/*
+	 * MSR_FS_BASE is in the guest-writable passthrough set below, so it
+	 * has to be saved here or a guest WRMSR to it survives the run and
+	 * corrupts the host's FS base.
+	 */
+	host->orig_fs_base = native_rdmsrq(MSR_FS_BASE);
+	host->orig_gs_base = native_rdmsrq(MSR_GS_BASE);
+	host->orig_kernel_gs_base = native_rdmsrq(MSR_KERNEL_GS_BASE);
+	host->orig_star = native_rdmsrq(MSR_STAR);
+	host->orig_lstar = native_rdmsrq(MSR_LSTAR);
+	host->orig_fmask = native_rdmsrq(MSR_SYSCALL_MASK);
+
+	/* Ensure Local APIC is software enabled */
+	apic_base = native_rdmsrq(MSR_IA32_APICBASE);
+	if (!(apic_base & MSR_IA32_APICBASE_ENABLE))
+		native_wrmsrq(MSR_IA32_APICBASE,
+			      apic_base | MSR_IA32_APICBASE_ENABLE);
+
+	/* Switch to self-contained Caretaker GDT, IDT, and TSS before CR3 switch */
+	kvm_x86_caretaker_load_desc(cxp->gdt, sizeof(cxp->gdt),
+				    caretaker_x86_idt, sizeof(caretaker_x86_idt),
+				    &cxp->tss);
+
+	/* Switch to preserved CR3 if specified */
+	sctx = cpu_preserved_get_stack_context();
+	if (sctx && sctx->session_pgd_pa)
+		cxp->host_cr3 = sctx->session_pgd_pa;
+	else if (!cxp->host_cr3 && x86_caretaker_pgd_pa)
+		cxp->host_cr3 = x86_caretaker_pgd_pa;
+	if (cxp->host_cr3 && __read_cr3() != cxp->host_cr3)
+		write_cr3(cxp->host_cr3);
+
+	raw_local_irq_disable();
+}
+
+static __caretaker_text void
+kvm_x86_caretaker_restore_host_state(const struct caretaker_x86_host_state *host,
+				     int pcpu)
+{
+	/*
+	 * Restore unconditionally.  These are all in the guest-writable
+	 * passthrough set, so skipping the write when the saved value happens
+	 * to be zero leaves the *guest's* value live in the host MSR.
+	 */
+	native_write_cr2(host->orig_cr2);
+	native_wrmsrq(MSR_FS_BASE, host->orig_fs_base);
+	native_wrmsrq(MSR_GS_BASE, host->orig_gs_base);
+	native_wrmsrq(MSR_KERNEL_GS_BASE, host->orig_kernel_gs_base);
+	native_wrmsrq(MSR_LSTAR, host->orig_lstar);
+	native_wrmsrq(MSR_STAR, host->orig_star);
+	native_wrmsrq(MSR_SYSCALL_MASK, host->orig_fmask);
+
+	if (cpu_is_preserved(pcpu))
+		arch_cpu_preserved_load_desc();
+	else if (host->orig_idt.size)
+		native_load_idt(&host->orig_idt);
+}
+
+__caretaker_text void
+kvm_x86_caretaker_update_msr(struct kvm_vcpu_arch_ser *state,
+			     u32 msr, u64 val)
+{
+	u32 i;
+
+	if (!state)
+		return;
+
+	for (i = 0; i < state->num_msrs; i++) {
+		if (state->msrs[i].index == msr) {
+			state->msrs[i].data = val;
+			return;
+		}
+	}
+}
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_x86_caretaker_update_msr);
+
+static __caretaker_text bool
+kvm_x86_caretaker_read_msr(const struct kvm_vcpu_arch_ser *state,
+			   u32 msr, u64 *val)
+{
+	u32 i;
+
+	if (!state)
+		return false;
+
+	for (i = 0; i < state->num_msrs; i++) {
+		if (state->msrs[i].index == msr) {
+			*val = state->msrs[i].data;
+			return true;
+		}
+	}
+	return false;
+}
+
+/*
+ * Capture the guest FPU registers into the LUO ABI buffer.
+ *
+ * The Caretaker runs the guest with the guest's FPU state live in hardware,
+ * restoring it via XRSTOR64 at the start of each quantum and saving it via
+ * XSAVE64 at the end of each quantum and upon detach.
+ *
+ * XSAVE -- as opposed to XSAVES -- writes the standard, non-compacted layout,
+ * which is bit-for-bit the uAPI struct kvm_xsave layout that the incoming
+ * kernel feeds to fpu_copy_uabi_to_guest_fpstate().  No format conversion is
+ * needed and the ABI stays uAPI.
+ *
+ * The requested-feature bitmap comes from the XCR0 recorded at preserve time
+ * rather than from XGETBV, because XGETBV requires CR4.OSXSAVE and the guest
+ * is free to clear it.  The recorded value cannot have gone stale: the
+ * Caretaker never emulates XSETBV, so the guest cannot change XCR0 while it
+ * runs here.
+ *
+ * The destination cannot overflow: kvm_arch_vcpu_luo_preserve() refuses the
+ * preserve when guest_fpu.uabi_size exceeds sizeof(struct kvm_xsave), and
+ * RFBM is a subset of guest_supported_xcr0, which is what uabi_size sizes.
+ */
+__caretaker_text static void
+caretaker_save_guest_fpu(struct caretaker_x86_page *cxp,
+			 struct kvm_vcpu_arch_ser *state)
+{
+	union fpregs_state *xstate = (union fpregs_state *)state->xsave.region;
+	u64 rfbm = state->xcrs.xcrs[0].value | XFEATURE_MASK_FPSSE;
+
+	if (!cxp->save_guest_fpu)
+		return;
+
+	if (native_read_cr0() & X86_CR0_TS)
+		asm volatile("clts" : : : "memory");
+
+	/*
+	 * XSAVE leaves XSTATE_BV bits for components outside RFBM untouched,
+	 * so the preserve-time header would survive and advertise stale
+	 * component data.  Clear it and let XSAVE set only what it writes.
+	 */
+	cpu_preserved_memset(&xstate->xsave.header, 0,
+			     sizeof(xstate->xsave.header));
+
+	asm volatile("1: xsave64 %[buf]\n\t"
+		     "2:\n\t"
+		     _ASM_EXTABLE(1b, 2b)
+		     : [buf] "+m" (*xstate)
+		     : "a" ((u32)rfbm), "d" ((u32)(rfbm >> 32))
+		     : "memory");
+}
+
+__caretaker_text void
+kvm_x86_caretaker_detach_serialize_common(struct caretaker_x86_page *cxp,
+					  struct kvm_vcpu_arch_ser *state)
+{
+	if (!cxp || !state)
+		return;
+
+	state->regs.rax = cxp->rax;
+	state->regs.rbx = cxp->rbx;
+	state->regs.rcx = cxp->rcx;
+	state->regs.rdx = cxp->rdx;
+	state->regs.rsi = cxp->rsi;
+	state->regs.rdi = cxp->rdi;
+	state->regs.rbp = cxp->rbp;
+	state->regs.r8  = cxp->r8;
+	state->regs.r9  = cxp->r9;
+	state->regs.r10 = cxp->r10;
+	state->regs.r11 = cxp->r11;
+	state->regs.r12 = cxp->r12;
+	state->regs.r13 = cxp->r13;
+	state->regs.r14 = cxp->r14;
+	state->regs.r15 = cxp->r15;
+
+	if (cxp->last_exit_rip)
+		state->regs.rip = cxp->last_exit_rip;
+	if (cxp->last_exit_rsp)
+		state->regs.rsp = cxp->last_exit_rsp;
+	if (cxp->last_exit_rflags)
+		state->regs.rflags = cxp->last_exit_rflags;
+
+	if (cxp->cr0)
+		state->sregs.cr0 = cxp->cr0;
+	if (cxp->cr3)
+		state->sregs.cr3 = cxp->cr3;
+	if (cxp->cr4)
+		state->sregs.cr4 = cxp->cr4;
+	if (cxp->efer)
+		state->sregs.efer = cxp->efer;
+
+	state->events.exception.injected = 0;
+	state->events.interrupt.injected = 0;
+
+	caretaker_save_guest_fpu(cxp, state);
+}
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_x86_caretaker_detach_serialize_common);
+
+void kvm_x86_caretaker_sync_vcpu_common(struct kvm_vcpu *vcpu)
+{
+	kvm_register_mark_dirty(vcpu, VCPU_REG_CR3);
+	kvm_clear_interrupt_queue(vcpu);
+	kvm_clear_exception_queue(vcpu);
+
+	vcpu->cpu = -1;
+	kvm_make_request(KVM_REQ_LOAD_MMU_PGD, vcpu);
+	kvm_make_request(KVM_REQ_TLB_FLUSH_CURRENT, vcpu);
+	kvm_make_request(KVM_REQ_RECALC_INTERCEPTS, vcpu);
+}
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_x86_caretaker_sync_vcpu_common);
+
+static __caretaker_text void kvm_caretaker_emulate_cpuid(u64 *rax,
+							 u64 *rbx,
+							 u64 *rcx,
+							 u64 *rdx)
+{
+	unsigned int a = (unsigned int)*rax;
+	unsigned int b = (unsigned int)*rbx;
+	unsigned int c = (unsigned int)*rcx;
+	unsigned int d = (unsigned int)*rdx;
+
+	asm volatile("cpuid"
+		     : "=a" (a), "=b" (b), "=c" (c), "=d" (d)
+		     : "0" (a), "2" (c));
+
+	*rax = a;
+	*rbx = b;
+	*rcx = c;
+	*rdx = d;
+}
+
+static __caretaker_text bool kvm_caretaker_emulate_msr(struct caretaker_x86_page *cxp,
+						       u32 msr, bool write,
+						       u64 *rax,
+						       u64 *rdx)
+{
+	bool x2apic = msr >= APIC_BASE_MSR &&
+		      msr < APIC_BASE_MSR + X2APIC_MSR_COUNT;
+	u32 apic_id = cxp ? cxp->abi.cb.vcpu_id : 0;
+	u64 val;
+
+	if (write) {
+		val = (u32)(*rax) | ((*rdx) << 32);
+
+		/*
+		 * Guest x2APIC writes are not emulated.  ICR would send an
+		 * IPI, TMICT would arm the APIC timer, and the LVT and TPR
+		 * registers reprogram delivery.  The caretaker implements
+		 * none of that, so absorbing the write promises the guest an
+		 * interrupt that will never arrive -- it wedges rather than
+		 * stalls, and it cannot tell the difference.
+		 *
+		 * Park instead, and let the incoming kernel's full KVM apply
+		 * the write to the emulated LAPIC when it reclaims the vCPU.
+		 *
+		 * Exception: APIC_EOI (0x80b).  If a vCPU was caught inside an
+		 * interrupt handler when detached, acknowledging EOI lets it
+		 * finish the ISR and IRETQ back to user space; kvm_luo clears
+		 * APIC_ISR on retrieve anyway.
+		 */
+		if (x2apic) {
+			if (msr == APIC_BASE_MSR + (APIC_EOI >> 4))
+				return true;
+			return false;
+		}
+
+		switch (msr) {
+		case MSR_IA32_SPEC_CTRL:
+		case MSR_IA32_PRED_CMD:
+			/*
+			 * The guest is arming a speculation mitigation
+			 * (IBRS/STIBP/SSBD, or an IBPB barrier).  The
+			 * caretaker does not apply these, so acknowledging
+			 * the write would leave the guest believing it is
+			 * protected when it is not -- a security downgrade
+			 * the guest cannot observe.
+			 *
+			 * Refuse the exit instead: the vCPU parks here and
+			 * the incoming kernel's KVM applies the write for
+			 * real when it reclaims the vCPU.
+			 */
+			return false;
+		case MSR_IA32_TSC_DEADLINE:
+			/*
+			 * Record the guest's next timer deadline in preserved
+			 * arch_state so full KVM restores and arms it upon
+			 * reclaiming the vCPU, while allowing a guest caught
+			 * in its timer ISR to return to user space.
+			 */
+			if (cxp && cxp->arch_state)
+				kvm_x86_caretaker_update_msr(cxp->arch_state,
+							     MSR_IA32_TSC_DEADLINE,
+							     val);
+			return true;
+		case MSR_IA32_TSC:
+		case MSR_IA32_TSC_ADJUST:
+			/*
+			 * Discarding these silently rewrites the guest's view of
+			 * time.
+			 */
+			return false;
+		case MSR_KERNEL_GS_BASE:
+			/* Also cached, so the read side can answer without an rdmsr. */
+			if (cxp)
+				cxp->kernel_gs_base = val;
+			fallthrough;
+		case MSR_FS_BASE:
+		case MSR_GS_BASE:
+		case MSR_LSTAR:
+		case MSR_STAR:
+		case MSR_SYSCALL_MASK:
+			native_wrmsrq(msr, val);
+			return true;
+		case MSR_IA32_APICBASE:
+			/*
+			 * This used to be passed through to native_wrmsrq(),
+			 * which let the guest relocate or disable the *physical*
+			 * APIC of the CPU the caretaker is running on.  Nothing
+			 * saved or restored it around the run, so the damage
+			 * outlived the guest: on the "staying in this kernel"
+			 * path there is no INIT-SIPI-SIPI to clean up after.
+			 *
+			 * APIC base is host state here.  Refuse the write.
+			 */
+			return false;
+		}
+		return false;
+	}
+
+	if (x2apic) {
+		switch ((msr - APIC_BASE_MSR) << 4) {
+		case APIC_ID:
+			val = apic_id;
+			break;
+		case APIC_LVR:
+			val = CARETAKER_APIC_LVR;
+			break;
+		case APIC_SPIV:
+			val = APIC_SPIV_APIC_ENABLED | APIC_VECTOR_MASK;
+			break;
+		case APIC_LDR:
+			val = ((apic_id >> 4) << 16) | (1U << (apic_id & 0xf));
+			break;
+		default:
+			/*
+			 * ICR, IRR, ISR, TMCCT and friends.  Zero reads as
+			 * "nothing pending" or "timer already expired", which
+			 * the guest cannot distinguish from the truth.  The four
+			 * cases above are answered because they are static
+			 * identity registers whose values really are known.
+			 */
+			return false;
+		}
+		goto out;
+	}
+
+	switch (msr) {
+	case MSR_IA32_SPEC_CTRL:
+		/*
+		 * Returning 0 here would tell the guest its speculation
+		 * mitigations are disabled, which is both wrong and
+		 * unobservable.  Park instead; see the write path above.
+		 */
+		return false;
+	case MSR_IA32_TSC:
+		val = rdtsc();
+		break;
+	case MSR_IA32_TSC_DEADLINE:
+		if (cxp && kvm_x86_caretaker_read_msr(cxp->arch_state,
+						      MSR_IA32_TSC_DEADLINE,
+						      &val))
+			break;
+		return false;
+	case MSR_IA32_TSC_ADJUST:
+		return false;
+	case MSR_KERNEL_GS_BASE:
+		if (cxp && cxp->kernel_gs_base)
+			val = cxp->kernel_gs_base;
+		else
+			val = native_rdmsrq(MSR_KERNEL_GS_BASE);
+		break;
+	case MSR_IA32_APICBASE:
+		val = native_rdmsrq(MSR_IA32_APICBASE);
+		if (!val)
+			val = APIC_DEFAULT_PHYS_BASE | MSR_IA32_APICBASE_ENABLE;
+		if (cxp && apic_id == 0)
+			val |= MSR_IA32_APICBASE_BSP;
+		else
+			val &= ~MSR_IA32_APICBASE_BSP;
+		break;
+	case MSR_FS_BASE:
+	case MSR_GS_BASE:
+	case MSR_LSTAR:
+	case MSR_STAR:
+	case MSR_SYSCALL_MASK:
+		val = native_rdmsrq(msr);
+		break;
+	default:
+		return false;
+	}
+
+out:
+	*rax = (u32)val;
+	*rdx = (u32)(val >> 32);
+	return true;
+}
+
+static bool __cpu_preserved_text
+kvm_x86_caretaker_emulate_uart8250(struct caretaker_uart *uart,
+				   u16 port, int in, int size,
+				   unsigned long *rax)
+{
+	u8 offset;
+
+	if (port < COM1_PORT_BASE || port > COM1_PORT_END)
+		return false;
+
+	offset = port - COM1_PORT_BASE;
+
+	if (in) {
+		unsigned long val = 0;
+
+		switch (offset) {
+		case UART_RX:
+			val = (uart && (uart->lcr & UART_LCR_DLAB)) ? uart->dll : 0;
+			break;
+		case UART_IER:
+			val = (uart && (uart->lcr & UART_LCR_DLAB)) ? uart->dlm :
+				(uart ? uart->ier : 0);
+			break;
+		case UART_IIR:
+			val = UART_IIR_NO_INT;
+			break;
+		case UART_LCR:
+			val = uart ? uart->lcr : UART_LCR_WLEN8;
+			break;
+		case UART_MCR:
+			val = uart ? uart->mcr : (UART_MCR_DTR | UART_MCR_RTS);
+			break;
+		case UART_LSR:
+			val = UART_LSR_TEMT | UART_LSR_THRE;
+			break;
+		case UART_MSR:
+			val = UART_MSR_DCD | UART_MSR_DSR | UART_MSR_CTS;
+			break;
+		case UART_SCR:
+			val = uart ? uart->scr : 0;
+			break;
+		}
+
+		if (size < (int)sizeof(unsigned long)) {
+			unsigned long mask = (1UL << (size * 8)) - 1;
+			*rax = (*rax & ~mask) | (val & mask);
+		} else {
+			*rax = val;
+		}
+	} else {
+		u8 out_val = (u8)*rax;
+
+		if (uart) {
+			switch (offset) {
+			case UART_TX:
+				if (uart->lcr & UART_LCR_DLAB)
+					uart->dll = out_val;
+				break;
+			case UART_IER:
+				if (uart->lcr & UART_LCR_DLAB)
+					uart->dlm = out_val;
+				else
+					uart->ier = out_val;
+				break;
+			case UART_LCR:
+				uart->lcr = out_val;
+				break;
+			case UART_MCR:
+				uart->mcr = out_val;
+				break;
+			case UART_SCR:
+				uart->scr = out_val;
+				break;
+			}
+		}
+	}
+
+	return true;
+}
+STACK_FRAME_NON_STANDARD(kvm_x86_caretaker_emulate_uart8250);
+
+__caretaker_text bool
+kvm_x86_caretaker_handle_exit(void *data, struct kvm_caretaker_exit *exit)
+{
+	struct caretaker_x86_page *cxp = data;
+	bool handled = false;
+
+	if (exit->type == KVM_CARETAKER_EXIT_CROSS_VCPU) {
+		/*
+		 * x86 has no cross-vCPU emulation.  The decoders route
+		 * VMCALL, APIC_ACCESS, APIC_WRITE, EOI_INDUCED and
+		 * INTERRUPT_WINDOW here, and every one of them has a
+		 * guest-visible effect the Caretaker cannot produce: a
+		 * hypercall it cannot service, an APIC register write it
+		 * cannot apply, an EOI it cannot retire, an IPI it cannot
+		 * deliver to a vCPU parked on another core.
+		 *
+		 * Returning true absorbed all of it.  Worse, nothing
+		 * advanced RIP afterwards, so VMCALL re-executed forever.
+		 *
+		 * Stall instead.  The vCPU parks on the instruction and the
+		 * incoming kernel's full KVM emulates it properly.  arm64
+		 * does handle its CROSS_VCPU case (SGI delivery) and keeps
+		 * returning true.
+		 */
+		return false;
+	}
+
+	switch ((int)exit->type) {
+	case KVM_CARETAKER_EXIT_CONSOLE: {
+		unsigned long *target = exit->mmio_io.val_ptr ?
+					(unsigned long *)exit->mmio_io.val_ptr :
+					(unsigned long *)&exit->mmio_io.val;
+
+		handled = kvm_x86_caretaker_emulate_uart8250(&cxp->uart,
+							     (u16)exit->mmio_io.addr,
+							     !exit->mmio_io.is_write,
+							     exit->mmio_io.size,
+							     target);
+		break;
+	}
+	case KVM_CARETAKER_EXIT_CPUID:
+		kvm_caretaker_emulate_cpuid(&cxp->rax, &cxp->rbx, &cxp->rcx, &cxp->rdx);
+		handled = true;
+		break;
+	case KVM_CARETAKER_EXIT_MSR:
+		handled = kvm_caretaker_emulate_msr(cxp, exit->msr.msr, exit->msr.is_write,
+						    &cxp->rax, &cxp->rdx);
+		break;
+	case KVM_CARETAKER_EXIT_RDTSC: {
+		u64 tsc = rdtsc();
+
+		cxp->rax = (u32)tsc;
+		cxp->rdx = (u32)(tsc >> 32);
+		handled = true;
+		break;
+	}
+	case KVM_CARETAKER_EXIT_INSN_STEP:
+		handled = true;
+		break;
+	case KVM_CARETAKER_EXIT_ARCH:
+	default:
+		/*
+		 * Nothing above recognised this exit, so nothing emulated it.
+		 * Advancing RIP here would step over an instruction whose
+		 * architectural effect never happened (MOV to CRn, XSETBV,
+		 * INVLPG, WBINVD, RDPMC, ...), leaving the guest running on
+		 * silently wrong state with no way to detect it.
+		 *
+		 * Report the exit as unhandled instead.  The caretaker run
+		 * loop stops re-entering the guest and the vCPU stays parked
+		 * on this instruction until the incoming kernel reclaims it
+		 * and full KVM emulates the exit properly.
+		 */
+		return false;
+	}
+
+	if (handled)
+		exit->rip += exit->insn_len;
+
+	return handled;
+}
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_x86_caretaker_handle_exit);
+
 __caretaker_text void kvm_x86_caretaker_arm_timer(u64 deadline_ticks)
 {
 	if (!deadline_ticks)
-- 
2.55.0.1082.g2b9226bbc0-goog




More information about the kexec mailing list