[Date Prev][Date Next][Thread Prev][Thread Next][Date Index][Thread Index]

[PATCH v1 4/5] x86/nestedsvm: update vmcb correctly


  • To: xen-devel@xxxxxxxxxxxxxxxxxxxx
  • From: chunjie.zhu@xxxxxxxxxx
  • Date: Wed, 23 Sep 2026 17:20:44 +0800
  • Arc-authentication-results: i=1; mx.microsoft.com 1; spf=pass smtp.mailfrom=citrix.com; dmarc=pass action=none header.from=citrix.com; dkim=pass header.d=citrix.com; arc=none
  • Arc-message-signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=microsoft.com; s=arcselector10001; h=From:Date:Subject:Message-ID:Content-Type:MIME-Version:X-MS-Exchange-AntiSpam-MessageData-ChunkCount:X-MS-Exchange-AntiSpam-MessageData-0:X-MS-Exchange-AntiSpam-MessageData-1; bh=c0G5DFHvfT3OxxH3EakjBBKP8uv26iIDoAqt94XNIBo=; b=roy0obYufjH+LX15/CI78RIdU4iOriNQmSe8Cf/lPDUoQO7rXgOizUFojNi1qJunxfuqfTz7VCeuVyuwW3iYVykGNghJAJazXH/X6VklwF7fO3E2wWoPgrCtTAta1eOe7Khqwry0ADvccBXMuxS3Mk6p1yJcQKNOs8lWVIT9E0YrnwzOzWoyayZXYGtohdxR3TCuEEN/X7md+LVQSoX09f2hAjjpBN33ZXRw1hme4mgSoWQV7Pk4953Wpcv6QivzjogufiG003e+CBn7TeXVEWm9l2PEnNFLMfZ5WaBA+am8VLLdi2LF1cAGgWyaPGfyxoNF7MhI5D73+rndmyURmg==
  • Arc-seal: i=1; a=rsa-sha256; s=arcselector10001; d=microsoft.com; cv=none; b=oB8xzTn5m48EtYiUi9N+tYqbbTv8e7uzFakRrSIWtmiLG/K6IoDp7Z31zCGYdYSPN7rnWY7TWSdl1pjngaHyyzuGzvedVa3mgOrjZaDDCgIdnqCNL/iyRbaSh8gNlNZ0e66SRH2gxJyxUuQBLbwqeWcMLncMQsBdZlCr8Czns/FAHhfQpUbxf6H5CreY18aYw9/2rUVoQfUASUkDFuZyjCf7TFnMLCARnYWn3Y27dHy8ZSrXY5IFj2EWQzbKMI9j38bR2wItElJSqjG4tF81NWK9gOYeFL+MPa3Gc5k+49CI/C4P6xEwB7P7boagiPQg3LFLckMCcjOaGmjnVnDh6w==
  • Authentication-results: eu.smtp.expurgate.cloud; dkim=pass header.s=selector1 header.d=citrix.com header.i="@citrix.com" header.h="From:Date:Subject:Message-Id:Content-Type:MIME-Version:X-MS-Exchange-SenderADCheck"
  • Authentication-results: dkim=none (message not signed) header.d=none;dmarc=none action=none header.from=citrix.com;
  • Cc: jbeulich@xxxxxxxx, andrew.cooper3@xxxxxxxxxx, roger@xxxxxxxxxxxxxx, jason.andryuk@xxxxxxx, teddy.astie@xxxxxxxxxx, Chunjie Zhu <chunjie.zhu@xxxxxxxxxx>
  • Delivery-date: Wed, 23 Sep 2026 09:21:39 +0000
  • List-id: Xen developer discussion <xen-devel.lists.xenproject.org>

From: Chunjie Zhu <chunjie.zhu@xxxxxxxxxx>

Signed-off-by: Chunjie Zhu <chunjie.zhu@xxxxxxxxxx>
---
 xen/arch/x86/hvm/svm/nestedsvm.c         | 100 ++++++++++++++++-------
 xen/arch/x86/hvm/svm/svm.c               |   4 +
 xen/arch/x86/hvm/svm/vmcb.c              |   3 +
 xen/arch/x86/hvm/svm/vmcb.h              |   6 ++
 xen/arch/x86/include/asm/hvm/svm-types.h |   2 +-
 xen/arch/x86/include/asm/msr-index.h     |   1 +
 6 files changed, 87 insertions(+), 29 deletions(-)

diff --git a/xen/arch/x86/hvm/svm/nestedsvm.c b/xen/arch/x86/hvm/svm/nestedsvm.c
index 9e5e804a5b00..82e01e513e69 100644
--- a/xen/arch/x86/hvm/svm/nestedsvm.c
+++ b/xen/arch/x86/hvm/svm/nestedsvm.c
@@ -146,16 +146,15 @@ int cf_check nsvm_vcpu_reset(struct vcpu *v)
     svm->ns_exception_intercepts = 0;
     svm->ns_general1_intercepts = 0;
     svm->ns_general2_intercepts = 0;
+    svm->ns_general3_intercepts = 0;
 
     svm->ns_hap_enabled = 0;
     svm->ns_vmcb_guestcr3 = 0;
     svm->ns_vmcb_hostcr3 = 0;
     svm->ns_asid = 0;
     svm->ns_hostflags.bytes = 0;
-
     svm->ns_iomap = NULL;
 
-    svm->ns_gif = 1;
     return 0;
 }
 
@@ -416,6 +415,7 @@ static int nsvm_vmcb_prepare4vmrun(struct vcpu *v, struct 
cpu_user_regs *regs)
         svm->ns_exception_intercepts = ns_vmcb->_exception_intercepts;
         svm->ns_general1_intercepts = ns_vmcb->_general1_intercepts;
         svm->ns_general2_intercepts = ns_vmcb->_general2_intercepts;
+        svm->ns_general3_intercepts = ns_vmcb->_general3_intercepts;
     }
 
     /* We could track the cleanbits of the n1vmcb from
@@ -444,6 +444,8 @@ static int nsvm_vmcb_prepare4vmrun(struct vcpu *v, struct 
cpu_user_regs *regs)
         n1vmcb->_general1_intercepts | ns_vmcb->_general1_intercepts;
     n2vmcb->_general2_intercepts =
         n1vmcb->_general2_intercepts | ns_vmcb->_general2_intercepts;
+    n2vmcb->_general3_intercepts =
+        n1vmcb->_general3_intercepts | ns_vmcb->_general3_intercepts;
 
     /* Nested Pause Filter */
     if ( ns_vmcb->_general1_intercepts & GENERAL1_INTERCEPT_PAUSE )
@@ -472,6 +474,17 @@ static int nsvm_vmcb_prepare4vmrun(struct vcpu *v, struct 
cpu_user_regs *regs)
         n2vmcb->_vintr.fields.intr_masking = 1;
     }
 
+    /*
+     * If L1 granted its L2 vGIF, the copy above already carried the enable
+     * and the GIF through.  Otherwise seed it set, as a VMRUN of a guest
+     * without vGIF starts with GIF set (APM vol.2 15.5).
+     */
+    if ( !ns_vmcb->_vintr.fields.vgif_enable )
+    {
+        n2vmcb->_vintr.fields.vgif_enable = 1;
+        n2vmcb->_vintr.fields.vgif = 1;
+    }
+
     /* Interrupt state */
     n2vmcb->int_stat = ns_vmcb->int_stat;
 
@@ -487,7 +500,8 @@ static int nsvm_vmcb_prepare4vmrun(struct vcpu *v, struct 
cpu_user_regs *regs)
     n2vmcb->virt_ext.bytes =
         n1vmcb->virt_ext.bytes | ns_vmcb->virt_ext.bytes;
 
-    /* NextRIP - only evaluated on #VMEXIT. */
+    /* NextRIP is consumed by VMRUN as the return address for injected soft 
events. */
+    n2vmcb->nextrip = ns_vmcb->nextrip;
 
     /*
      * VMCB Save State Area
@@ -652,10 +666,12 @@ nsvm_vcpu_vmentry(struct vcpu *v, struct cpu_user_regs 
*regs,
     int ret;
     struct nestedvcpu *nv = &vcpu_nestedhvm(v);
     struct nestedsvm *svm = &vcpu_nestedsvm(v);
-    struct vmcb_struct *ns_vmcb;
+    struct vmcb_struct *ns_vmcb, *n1vmcb;
 
     ns_vmcb = nv->nv_vvmcx;
+    n1vmcb = nv->nv_n1vmcx;
     ASSERT(ns_vmcb != NULL);
+    ASSERT(n1vmcb != NULL);
     ASSERT(nv->nv_n2vmcx != NULL);
     ASSERT(nv->nv_n2vmcx_pa != INVALID_PADDR);
 
@@ -700,7 +716,8 @@ nsvm_vcpu_vmentry(struct vcpu *v, struct cpu_user_regs 
*regs,
         return ret;
     }
 
-    svm->ns_gif = 1;
+    ASSERT(n1vmcb->_vintr.fields.vgif_enable);
+    n1vmcb->_vintr.fields.vgif = 1;
     return 0;
 }
 
@@ -756,7 +773,7 @@ nsvm_vcpu_vmexit_inject(struct vcpu *v, struct 
cpu_user_regs *regs,
     struct vmcb_struct *vmcb = v->arch.hvm.svm.vmcb;
 
     ASSERT(vmcb->_vintr.fields.vgif_enable);
-        vmcb->_vintr.fields.vgif = 0;
+    vmcb->_vintr.fields.vgif = 0;
 
     ns_vmcb = nv->nv_vvmcx;
 
@@ -939,12 +956,18 @@ nsvm_vmcb_guest_intercepts_exitcode(struct vcpu *v,
             break;
         return 0;
 
-    case VMEXIT_VMRUN ... VMEXIT_XSETBV:
+    case VMEXIT_VMRUN ... VMEXIT_RDPRU:
         exit_bits = 1ULL << (exitcode - VMEXIT_VMRUN);
         if ( svm->ns_general2_intercepts & exit_bits )
             break;
         return 0;
 
+    case VMEXIT_INVLPGB ... VMEXIT_IDLE_HLT:
+        exit_bits = 1ULL << (exitcode - VMEXIT_INVLPGB);
+        if ( svm->ns_general3_intercepts & exit_bits )
+            break;
+        return 0;
+
     case VMEXIT_NPF:
         if ( nestedhvm_paging_mode_hap(v) )
             break;
@@ -1026,10 +1049,15 @@ nsvm_vmcb_prepare4vmexit(struct vcpu *v, struct 
cpu_user_regs *regs)
     /* TLB control */
     ns_vmcb->tlb_control = 0;
 
-    /* Virtual Interrupts */
-    ns_vmcb->_vintr = n2vmcb->_vintr;
-    if ( !svm->ns_hostflags.fields.vintrmask )
-        ns_vmcb->_vintr.fields.intr_masking = 0;
+    /*
+     * Virtual Interrupts.  #VMEXIT writes back V_TPR and V_IRQ (APM vol.2
+     * 15.6); the remaining fields do not change.  Copy vGIF back only if
+     * enabled in L1.
+     */
+    ns_vmcb->_vintr.fields.tpr = n2vmcb->_vintr.fields.tpr;
+    ns_vmcb->_vintr.fields.irq = n2vmcb->_vintr.fields.irq;
+    if ( ns_vmcb->_vintr.fields.vgif_enable )
+        ns_vmcb->_vintr.fields.vgif = n2vmcb->_vintr.fields.vgif;
 
     /* Interrupt state */
     ns_vmcb->int_stat = n2vmcb->int_stat;
@@ -1437,19 +1465,30 @@ nestedsvm_vcpu_interrupt(struct vcpu *v, const struct 
hvm_intack intack)
 bool
 nestedsvm_gif_isset(struct vcpu *v)
 {
-    struct nestedsvm *svm = &vcpu_nestedsvm(v);
-    struct vmcb_struct *vmcb = v->arch.hvm.svm.vmcb;
+    struct nestedvcpu *nv = &vcpu_nestedhvm(v);
+    struct vmcb_struct *n1vmcb = nv->nv_n1vmcx;
+    struct vmcb_struct *n2vmcb = nv->nv_n2vmcx;
+    struct vmcb_struct *ns_vmcb = nv->nv_vvmcx;
 
-    /* get the vmcb gif value if using vgif */
-    if ( vmcb->_vintr.fields.vgif_enable )
-        return vmcb->_vintr.fields.vgif;
-    else
-        return svm->ns_gif;
+    /*
+     * An L2 with no vGIF of its own shares L1's GIF.  L0 keeps that shared
+     * GIF in the L2 shadow, to keep an L2 CLGI off the host's.
+     */
+    if ( nestedhvm_vcpu_in_guestmode(v) &&
+         !ns_vmcb->_vintr.fields.vgif_enable )
+        return n2vmcb->_vintr.fields.vgif;
+
+    /*
+     * The VMCB virtual GIF is authoritative.
+     */
+    return n1vmcb->_vintr.fields.vgif;
 }
 
 void svm_vmexit_do_stgi(struct cpu_user_regs *regs, struct vcpu *v)
 {
+    struct vmcb_struct *vmcb = v->arch.hvm.svm.vmcb;
     unsigned int inst_len;
+    vintr_t intr;
 
     /*
      * STGI doesn't require SVME to be set to be used.  See AMD APM vol
@@ -1464,7 +1503,9 @@ void svm_vmexit_do_stgi(struct cpu_user_regs *regs, 
struct vcpu *v)
     if ( (inst_len = svm_get_insn_len(v, INSTR_STGI)) == 0 )
         return;
 
-    vcpu_nestedsvm(v).ns_gif = 1;
+    intr = vmcb_get_vintr(vmcb);
+    intr.fields.vgif = 1;
+    vmcb_set_vintr(vmcb, intr);
 
     __update_guest_eip(regs, inst_len);
 }
@@ -1485,10 +1526,9 @@ void svm_vmexit_do_clgi(struct cpu_user_regs *regs, 
struct vcpu *v)
     if ( (inst_len = svm_get_insn_len(v, INSTR_CLGI)) == 0 )
         return;
 
-    vcpu_nestedsvm(v).ns_gif = 0;
-
     /* After a CLGI no interrupts should come */
     intr = vmcb_get_vintr(vmcb);
+    intr.fields.vgif = 0;
     intr.fields.irq = 0;
     general1_intercepts &= ~GENERAL1_INTERCEPT_VINTR;
     vmcb_set_vintr(vmcb, intr);
@@ -1504,10 +1544,17 @@ void svm_vmexit_do_clgi(struct cpu_user_regs *regs, 
struct vcpu *v)
 void svm_nested_features_on_efer_update(struct vcpu *v)
 {
     struct vmcb_struct *vmcb = v->arch.hvm.svm.vmcb;
-    struct nestedsvm *svm = &vcpu_nestedsvm(v);
     u32 general2_intercepts;
     vintr_t vintr;
 
+    /*
+     * These accelerate the L1 VMCB.  A world switch loads an EFER rather
+     * than writing one, and an L2 EFER write says nothing about L1; the L2
+     * shadow is configured in nsvm_vmcb_prepare4vmrun().
+     */
+    if ( nestedhvm_vmswitch_in_progress(v) || nestedhvm_vcpu_in_guestmode(v) )
+        return;
+
     /*
      * Need state for transfering the nested gif status so only write on
      * the hvm_vcpu EFER.SVME changing.
@@ -1515,7 +1562,6 @@ void svm_nested_features_on_efer_update(struct vcpu *v)
     if ( nsvm_efer_svm_enabled(v) )
     {
         if ( !vmcb->virt_ext.fields.vloadsave_enable &&
-             paging_mode_hap(v->domain) &&
              cpu_has_svm_vloadsave )
         {
             vmcb->virt_ext.fields.vloadsave_enable = 1;
@@ -1525,11 +1571,9 @@ void svm_nested_features_on_efer_update(struct vcpu *v)
             vmcb_set_general2_intercepts(vmcb, general2_intercepts);
         }
 
-        if ( !vmcb->_vintr.fields.vgif_enable &&
-             cpu_has_svm_vgif )
+        if ( !vmcb->_vintr.fields.vgif_enable )
         {
             vintr = vmcb_get_vintr(vmcb);
-            vintr.fields.vgif = svm->ns_gif;
             vintr.fields.vgif_enable = 1;
             vmcb_set_vintr(vmcb, vintr);
             general2_intercepts  = vmcb_get_general2_intercepts(vmcb);
@@ -1552,7 +1596,6 @@ void svm_nested_features_on_efer_update(struct vcpu *v)
         if ( vmcb->_vintr.fields.vgif_enable )
         {
             vintr = vmcb_get_vintr(vmcb);
-            svm->ns_gif = vintr.fields.vgif;
             vintr.fields.vgif_enable = 0;
             vmcb_set_vintr(vmcb, vintr);
             general2_intercepts  = vmcb_get_general2_intercepts(vmcb);
@@ -1574,5 +1617,6 @@ void __init start_nested_svm(struct hvm_function_table 
*hvm_function_table)
         cpu_has_svm_lbrv &&
         cpu_has_svm_nrips &&
         cpu_has_svm_flushbyasid &&
-        cpu_has_svm_decode;
+        cpu_has_svm_decode &&
+        cpu_has_svm_vgif;
 }
diff --git a/xen/arch/x86/hvm/svm/svm.c b/xen/arch/x86/hvm/svm/svm.c
index a8d258aa3980..074f09a0e0e9 100644
--- a/xen/arch/x86/hvm/svm/svm.c
+++ b/xen/arch/x86/hvm/svm/svm.c
@@ -1660,6 +1660,8 @@ static void svm_dr_access(struct vcpu *v, struct 
cpu_user_regs *regs)
 
     TRACE(TRC_HVM_DR_WRITE);
     __restore_debug_registers(vmcb, v);
+    if ( nestedhvm_enabled(v->domain) && nestedhvm_vcpu_in_guestmode(v) )
+        vmcb_set_dr_intercepts(v->arch.hvm.svm.vmcb, 0);
 }
 
 static int cf_check svm_msr_read_intercept(
@@ -1990,6 +1992,8 @@ static int cf_check svm_msr_write_intercept(
     case MSR_K8_SYSCFG:
     case MSR_K8_VM_CR:
     case MSR_AMD64_EX_CFG:
+    case MSR_K8_HWCR:
+    case MSR_K8_VM_IGNNE:
         /* ignore write. handle all bits as read-only. */
         break;
 
diff --git a/xen/arch/x86/hvm/svm/vmcb.c b/xen/arch/x86/hvm/svm/vmcb.c
index d069280a4da8..5354c4f1b85f 100644
--- a/xen/arch/x86/hvm/svm/vmcb.c
+++ b/xen/arch/x86/hvm/svm/vmcb.c
@@ -194,6 +194,9 @@ static int construct_vmcb(struct vcpu *v)
 
     vmcb->_vintr.fields.vnmi_enable = cpu_has_svm_vnmi;
 
+    /* GIF is set out of reset; with vGIF its value lives in the VMCB. */
+    vmcb->_vintr.fields.vgif = 1;
+
     return 0;
 }
 
diff --git a/xen/arch/x86/hvm/svm/vmcb.h b/xen/arch/x86/hvm/svm/vmcb.h
index 3760f71a8625..d098bf1d1355 100644
--- a/xen/arch/x86/hvm/svm/vmcb.h
+++ b/xen/arch/x86/hvm/svm/vmcb.h
@@ -288,7 +288,13 @@ enum VMEXIT_EXITCODE
     VMEXIT_MWAIT_CONDITIONAL= 140, /* 0x8c */
     VMEXIT_XSETBV           = 141, /* 0x8d */
     VMEXIT_RDPRU            = 142, /* 0x8e */
+    VMEXIT_INVLPGB          = 160, /* 0xa0 */
+    VMEXIT_INVLPGB_ILLEGAL  = 161, /* 0xa1 */
+    VMEXIT_INVPCID          = 162, /* 0xa2 */
+    VMEXIT_MCOMMIT          = 163, /* 0xa3 */
+    VMEXIT_TLBSYNC          = 164, /* 0xa4 */
     VMEXIT_BUS_LOCK         = 165, /* 0xa5 */
+    VMEXIT_IDLE_HLT         = 166, /* 0xa6 */
     /* Remember to also update VMEXIT_NPF_PERFC! */
     VMEXIT_NPF              = 1024, /* 0x400, nested paging fault */
     /* Remember to also update SVM_PERF_EXIT_REASON_SIZE! */
diff --git a/xen/arch/x86/include/asm/hvm/svm-types.h 
b/xen/arch/x86/include/asm/hvm/svm-types.h
index 256bf5b91402..24363a58ffcb 100644
--- a/xen/arch/x86/include/asm/hvm/svm-types.h
+++ b/xen/arch/x86/include/asm/hvm/svm-types.h
@@ -44,6 +44,7 @@ struct nestedsvm {
     uint32_t ns_exception_intercepts;
     uint32_t ns_general1_intercepts;
     uint32_t ns_general2_intercepts;
+    uint32_t ns_general3_intercepts;
 
     /* Cached real MSR permission bitmaps of the l2 guest */
     unsigned long *ns_cached_msrpm;
@@ -66,7 +67,6 @@ struct nestedsvm {
     uint64_t ns_vmcb_guestcr3, ns_vmcb_hostcr3;
     uint32_t ns_asid;
 
-    bool ns_gif;
     bool ns_hap_enabled;
 
     union {
diff --git a/xen/arch/x86/include/asm/msr-index.h 
b/xen/arch/x86/include/asm/msr-index.h
index ad1c6c97f8f7..beb858bf99bb 100644
--- a/xen/arch/x86/include/asm/msr-index.h
+++ b/xen/arch/x86/include/asm/msr-index.h
@@ -250,6 +250,7 @@
 #define MSR_K8_VM_CR                        _AC(0xc0010114, U)
 #define  VM_CR_INIT_REDIRECTION             (_AC(1, ULL) <<  1)
 #define  VM_CR_SVM_DISABLE                  (_AC(1, ULL) <<  4)
+#define MSR_K8_VM_IGNNE                     _AC(0xc0010115, U)
 
 #define MSR_VIRT_SPEC_CTRL                  _AC(0xc001011f, U) /* Layout 
matches MSR_SPEC_CTRL */
 
-- 
2.34.1




 


Rackspace

Lists.xenproject.org is hosted with RackSpace, monitoring our
servers 24x7x365 and backed by RackSpace's Fanatical Support®.