Kernel KVM virtualization development
 help / color / mirror / Atom feed
From: Sriram Nambakam <snambakam@linux.microsoft.com>
To: qemu-devel@nongnu.org
Cc: kvm@vger.kernel.org
Subject: [RFC PATCH v1 4/5] target/i386/kvm: run the secure plane in-kernel (Option B)
Date: Wed,  5 Aug 2026 04:04:31 -0700	[thread overview]
Message-ID: <20260805110432.25167-5-snambakam@linux.microsoft.com> (raw)
In-Reply-To: <20260805110432.25167-1-snambakam@linux.microsoft.com>

Move secure-plane execution into the kernel and out of QEMU. Plane vCPUs
are now created and initialised on their owning plane-0 CPU thread via
run_on_cpu() (plane_vcpu_create_cb / plane_vcpu_init_cb), and the
activate path leaves the secure vCPU stopped: it boots lazily inside KVM
on the first KVM_HC_VBS_VTL_CALL, which KVM now services itself.

Drop the userspace secure-world emulation that this replaces: the
dedicated vm_plane_vcpu_thread (with its 8250 emulation and per-plane
serial capture), vbs_apply_protection, vbs_handle_protect_memory,
vbs_handle_seal_kernel and the kvm_handle_hc_vbs_vtl_call dispatch.
Track each plane vCPU's owning CPU index (vcpu_cpu_index) so the
run_on_cpu callbacks target the right thread.

Signed-off-by: Sriram Nambakam <snambakam@linux.microsoft.com>
(cherry picked from commit cbecc9eda1514d200fefe2bb89f330be4f1ab1be)
---
 include/system/kvm_int.h |   1 +
 target/i386/kvm/kvm.c    | 536 ++++++++++++---------------------------
 2 files changed, 157 insertions(+), 380 deletions(-)

diff --git a/include/system/kvm_int.h b/include/system/kvm_int.h
index e7c9dc95a2..cde0a148e1 100644
--- a/include/system/kvm_int.h
+++ b/include/system/kvm_int.h
@@ -114,6 +114,7 @@ struct KVMPlane {
  * kvm_get_plane_fd(s, plane_id); only LVBS-specific state lives here. */
 struct kvm_vm_plane_state {
     int *vcpu_fds;
+    unsigned int *vcpu_cpu_index;  /* owning plane-0 CPU index per plane vCPU */
     unsigned int vcpu_count;
     uint64_t load_offset;
     uint64_t memory_size;
diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c
index 9d89e5cac6..5ac7962e61 100644
--- a/target/i386/kvm/kvm.c
+++ b/target/i386/kvm/kvm.c
@@ -6541,10 +6541,9 @@ static int kvm_handle_hc_map_gpa_range(X86CPU *cpu, struct kvm_run *run)
  *   off  0  u64  load_offset
  *   off  8  u64  memory_size
  *   off 16  u64  entry_point
- *   off 24  u32  vcpu_count
- *   off 28  u32  kernel_format
- *   off 32  char kernel[128]
- *   off 160 char cmdline[512]
+ *   off 24  u32  kernel_format
+ *   off 28  char kernel[128]
+ *   off 156 char cmdline[512]
  * ======================================================================== */
 
 #define VM_PLANE_CFG_STRIDE     672
@@ -6568,23 +6567,37 @@ static int kvm_handle_hc_map_gpa_range(X86CPU *cpu, struct kvm_run *run)
 #define CA_OFF_RESP_SIZE  16
 #define CA_OFF_BUFFER     20
 
-struct vm_plane_boot_ctx {
-    int vcpu_fd;
-    int vcpu_mmap_size;
-    unsigned int vcpu_idx;
-    uint64_t plane_id;
-    unsigned int *halted_count;
-    QemuMutex *mutex;
-    QemuCond *cond;
-    int result;
-    int log_fd;
-    QemuMutex wake_mutex;
-    QemuCond wake_cond;
-    bool halted;
-    bool kick;
-    bool stopped;
+/*
+ * Plane>0 vCPU creation and initialization must run on the OWNING plane-0
+ * vCPU's thread.  A plane>0 vCPU shares its plane-0 sibling's
+ * struct kvm_vcpu_common, including the single embedded preempt_notifier.
+ * KVM_CREATE_VCPU and KVM_SET_{S,}REGS / KVM_SET_MP_STATE all vcpu_load()
+ * the plane vCPU, which registers that shared notifier on the *calling*
+ * task.  If issued from the config/activate-hypercall vCPU's thread while a
+ * sibling is concurrently in KVM_RUN on another host CPU, the same
+ * hlist_node would be linked onto two tasks' preempt-notifier lists ->
+ * list corruption -> host hard lockup.  run_on_cpu() forces the sibling out
+ * of KVM_RUN (KVM_RUN returns, vcpu_put() unregisters the notifier) and
+ * runs the work on that sibling's own thread, so the notifier is owned by
+ * exactly one task.  For the current vCPU, run_on_cpu() runs inline.
+ */
+struct plane_vcpu_create_ctx {
+    KVMState *s;
+    unsigned int plane_id;
+    unsigned int vcpu_id;
+    int fd;                 /* out: vcpu fd, or -errno on failure */
 };
 
+static void plane_vcpu_create_cb(CPUState *cs, run_on_cpu_data data)
+{
+    struct plane_vcpu_create_ctx *ctx = data.host_ptr;
+    int fd;
+
+    fd = kvm_vm_plane_ioctl(ctx->s, ctx->plane_id, KVM_CREATE_VCPU,
+                            (void *)(uintptr_t)ctx->vcpu_id);
+    ctx->fd = (fd < 0) ? -errno : fd;
+}
+
 static int kvm_handle_hc_vm_planes_config(X86CPU *cpu, struct kvm_run *run)
 {
     uint64_t gpa = run->hypercall.args[0];
@@ -6645,7 +6658,7 @@ static int kvm_handle_hc_vm_planes_config(X86CPU *cpu, struct kvm_run *run)
     for (plane_id = 1; plane_id < plane_count; plane_id++) {
         uint64_t plane_gpa = gpa + (plane_id * VM_PLANE_CFG_STRIDE);
         uint64_t load_offset = 0, memory_size = 0, entry_point = 0;
-        uint32_t vcpu_count = 0;
+        uint32_t vcpu_count = plane0_vcpu_count;
         struct kvm_vm_plane_state *ps = &s->vm_planes[plane_id];
         char cmdline_buf[512];
         MemoryRegionSection section;
@@ -6655,29 +6668,24 @@ static int kvm_handle_hc_vm_planes_config(X86CPU *cpu, struct kvm_run *run)
         cpu_physical_memory_read(plane_gpa + 0,  &load_offset, 8);
         cpu_physical_memory_read(plane_gpa + 8,  &memory_size, 8);
         cpu_physical_memory_read(plane_gpa + 16, &entry_point, 8);
-        cpu_physical_memory_read(plane_gpa + 24, &vcpu_count,  4);
 
-        if (!memory_size || !vcpu_count) {
+        /*
+         * joergroedel plane model: a plane has exactly one vCPU per
+         * plane-0 vCPU — each is the sibling of a plane-0 vCPU sharing the
+         * same logical CPU.  The guest does not configure a count; QEMU
+         * mirrors the plane-0 vCPU set.
+         */
+        if (!memory_size) {
             error_report("vm_planes: plane %" PRIu64 " invalid "
-                         "(load_offset=0x%" PRIx64 " size=0x%" PRIx64
-                         " vcpus=%u)",
-                         plane_id, load_offset, memory_size, vcpu_count);
-            g_free(plane0_vcpu_ids);
-            run->hypercall.ret = -EINVAL;
-            return 0;
-        }
-
-        if (vcpu_count > plane0_vcpu_count) {
-            error_report("vm_planes: plane %" PRIu64 " requests %u vCPUs, "
-                         "but plane0 has %u", plane_id, vcpu_count,
-                         plane0_vcpu_count);
+                         "(load_offset=0x%" PRIx64 " size=0x%" PRIx64 ")",
+                         plane_id, load_offset, memory_size);
             g_free(plane0_vcpu_ids);
             run->hypercall.ret = -EINVAL;
             return 0;
         }
 
         memset(cmdline_buf, 0, sizeof(cmdline_buf));
-        cpu_physical_memory_read(plane_gpa + 160, cmdline_buf,
+        cpu_physical_memory_read(plane_gpa + 156, cmdline_buf,
                                  sizeof(cmdline_buf));
         cmdline_buf[sizeof(cmdline_buf) - 1] = '\0';
         memcpy(ps->cmdline, cmdline_buf, sizeof(ps->cmdline));
@@ -6714,24 +6722,46 @@ static int kvm_handle_hc_vm_planes_config(X86CPU *cpu, struct kvm_run *run)
         memory_region_unref(section.mr);
 
         ps->vcpu_fds = g_new0(int, vcpu_count);
+        ps->vcpu_cpu_index = g_new0(unsigned int, vcpu_count);
         for (i = 0; i < vcpu_count; i++) {
-            int vcpu_fd;
             unsigned int vcpu_id = plane0_vcpu_ids[i];
+            CPUState *target = qemu_get_cpu(vcpu_id);
+            struct plane_vcpu_create_ctx cctx = {
+                .s = s,
+                .plane_id = plane_id,
+                .vcpu_id = vcpu_id,
+                .fd = -EINVAL,
+            };
+
+            if (!target) {
+                error_report("vm_planes: plane %" PRIu64 " no CPU for id=%u",
+                             plane_id, vcpu_id);
+                close(plane_fd);
+                kvm_set_plane_fd(s, plane_id, -1);
+                g_free(plane0_vcpu_ids);
+                run->hypercall.ret = -EINVAL;
+                return 0;
+            }
 
-            vcpu_fd = kvm_vm_plane_ioctl(s, plane_id, KVM_CREATE_VCPU,
-                                         (void *)(uintptr_t)vcpu_id);
-            if (vcpu_fd < 0) {
+            /* Create on the owning CPU's thread; see plane_vcpu_create_cb. */
+            bql_lock();
+            run_on_cpu(target, plane_vcpu_create_cb,
+                       RUN_ON_CPU_HOST_PTR(&cctx));
+            bql_unlock();
+
+            if (cctx.fd < 0) {
                 error_report("vm_planes: KVM_CREATE_VCPU plane %" PRIu64
                              " vcpu %u failed: %s",
-                             plane_id, vcpu_id, strerror(errno));
+                             plane_id, vcpu_id, strerror(-cctx.fd));
                 close(plane_fd);
                 kvm_set_plane_fd(s, plane_id, -1);
                 g_free(plane0_vcpu_ids);
-                run->hypercall.ret = -errno;
+                run->hypercall.ret = cctx.fd;
                 return 0;
             }
 
-            ps->vcpu_fds[i] = vcpu_fd;
+            ps->vcpu_fds[i] = cctx.fd;
+            ps->vcpu_cpu_index[i] = vcpu_id;
         }
 
         ps->vcpu_count  = vcpu_count;
@@ -6757,10 +6787,6 @@ static int kvm_init_plane_vcpu(int vcpu_fd, uint64_t entry_addr,
 {
     struct kvm_regs regs = {};
     struct kvm_sregs sregs = {};
-    struct kvm_mp_state mp = {
-        .mp_state = is_bsp ? KVM_MP_STATE_RUNNABLE
-                           : KVM_MP_STATE_INIT_RECEIVED,
-    };
     int ret;
 
     sregs.cs.base = 0; sregs.cs.limit = 0xffffffff; sregs.cs.selector = 0x10;
@@ -6806,144 +6832,42 @@ static int kvm_init_plane_vcpu(int vcpu_fd, uint64_t entry_addr,
         return -errno;
     }
 
-    ret = ioctl(vcpu_fd, KVM_SET_MP_STATE, &mp);
-    if (ret < 0) {
-        error_report("vm_planes: KVM_SET_MP_STATE: %s", strerror(errno));
-        return -errno;
-    }
+    /*
+     * Do NOT issue KVM_SET_MP_STATE on a plane (plane_level > 0) vCPU fd:
+     * the VM-planes kernel only whitelists a subset of vCPU ioctls for
+     * plane vCPUs (KVM_SET_*REGS/SREGS/FPU/LAPIC/...), and KVM_SET_MP_STATE
+     * is intentionally excluded -> it returns -EINVAL.  The kernel already
+     * establishes the correct initial MP state at create time: the plane
+     * BSP (vcpu_id == bsp_vcpu_id) is left RUNNABLE, and APs are left
+     * UNINITIALIZED (wait-for-INIT), which is the correct power-on state.
+     * The plane guest's own SMP bringup INIT/SIPIs its APs via the
+     * in-kernel LAPIC, so no userspace MP-state poke is needed.
+     */
     return 0;
 }
 
-static void *vm_plane_vcpu_thread(void *arg)
-{
-    struct vm_plane_boot_ctx *ctx = arg;
-    struct kvm_run *kvm_run;
-    int ret;
-    bool boot_signaled = false;
-
-    kvm_run = mmap(NULL, ctx->vcpu_mmap_size, PROT_READ | PROT_WRITE,
-                   MAP_SHARED, ctx->vcpu_fd, 0);
-    if (kvm_run == MAP_FAILED) {
-        error_report("vm_planes: plane %" PRIu64 " vcpu %u: mmap failed: %s",
-                     ctx->plane_id, ctx->vcpu_idx, strerror(errno));
-        ctx->result = -errno;
-        qemu_mutex_lock(ctx->mutex);
-        (*ctx->halted_count)++;
-        qemu_cond_signal(ctx->cond);
-        qemu_mutex_unlock(ctx->mutex);
-        return NULL;
-    }
-
-    for (;;) {
-        ret = ioctl(ctx->vcpu_fd, KVM_RUN, 0);
-        if (ret < 0) {
-            if (errno == EINTR || errno == EAGAIN) {
-                continue;
-            }
-            error_report("vm_planes: plane %" PRIu64 " vcpu %u: KVM_RUN: %s",
-                         ctx->plane_id, ctx->vcpu_idx, strerror(errno));
-            ctx->result = -errno;
-            break;
-        }
-
-        switch (kvm_run->exit_reason) {
-        case KVM_EXIT_HLT:
-            if (!boot_signaled) {
-                boot_signaled = true;
-                qemu_mutex_lock(ctx->mutex);
-                (*ctx->halted_count)++;
-                qemu_cond_signal(ctx->cond);
-                qemu_mutex_unlock(ctx->mutex);
-            }
-            qemu_mutex_lock(&ctx->wake_mutex);
-            ctx->halted = true;
-            while (!ctx->kick && !ctx->stopped) {
-                qemu_cond_wait(&ctx->wake_cond, &ctx->wake_mutex);
-            }
-            ctx->halted = false;
-            ctx->kick = false;
-            if (ctx->stopped) {
-                qemu_mutex_unlock(&ctx->wake_mutex);
-                goto done;
-            }
-            qemu_mutex_unlock(&ctx->wake_mutex);
-            break;
-
-        case KVM_EXIT_IO: {
-            uint8_t *io_data = (uint8_t *)kvm_run + kvm_run->io.data_offset;
-            size_t io_size = kvm_run->io.size * kvm_run->io.count;
-            uint16_t port = kvm_run->io.port;
-
-            if (kvm_run->io.direction == KVM_EXIT_IO_OUT) {
-                if (port == 0x3f8 && ctx->log_fd >= 0) {
-                    ssize_t w = write(ctx->log_fd, io_data, io_size);
-                    (void)w;
-                }
-            } else {
-                memset(io_data, 0, io_size);
-                switch (port) {
-                case 0x3fa: memset(io_data, 0xc1, io_size); break;
-                case 0x3fb: memset(io_data, 0x03, io_size); break;
-                case 0x3fc: memset(io_data, 0x08, io_size); break;
-                case 0x3fd: memset(io_data, 0x60, io_size); break;
-                case 0x3fe: memset(io_data, 0xb0, io_size); break;
-                default: break;
-                }
-            }
-            usleep(100);
-            break;
-        }
-
-        case KVM_EXIT_MMIO:
-            usleep(100);
-            break;
-
-        case KVM_EXIT_SHUTDOWN: {
-            struct kvm_regs dbg = {};
-            ioctl(ctx->vcpu_fd, KVM_GET_REGS, &dbg);
-            error_report("vm_planes: plane %" PRIu64 " vcpu %u: shutdown "
-                         "RIP=0x%" PRIx64 " RSP=0x%" PRIx64,
-                         ctx->plane_id, ctx->vcpu_idx,
-                         (uint64_t)dbg.rip, (uint64_t)dbg.rsp);
-            ctx->result = -EFAULT;
-            goto done;
-        }
-
-        case KVM_EXIT_FAIL_ENTRY:
-            error_report("vm_planes: plane %" PRIu64 " vcpu %u: entry failure "
-                         "0x%" PRIx64, ctx->plane_id, ctx->vcpu_idx,
-                         (uint64_t)kvm_run->fail_entry.hardware_entry_failure_reason);
-            ctx->result = -EFAULT;
-            goto done;
-
-        case KVM_EXIT_INTERNAL_ERROR:
-            error_report("vm_planes: plane %" PRIu64 " vcpu %u: internal "
-                         "error %u", ctx->plane_id, ctx->vcpu_idx,
-                         kvm_run->internal.suberror);
-            ctx->result = -EFAULT;
-            goto done;
+/* Initialize a plane vCPU on its owning CPU's thread (see
+ * plane_vcpu_create_cb for why KVM_SET_*REGS / KVM_SET_MP_STATE, which
+ * vcpu_load() the shared common, must not race the running sibling). */
+struct plane_vcpu_init_ctx {
+    int vcpu_fd;
+    uint64_t entry_addr;
+    uint64_t stack_addr;
+    uint64_t zero_page_gpa;
+    uint64_t page_table_gpa;
+    uint64_t gdt_gpa;
+    bool is_bsp;
+    int ret;                /* out: 0 on success, -errno on failure */
+};
 
-        default:
-            error_report("vm_planes: plane %" PRIu64 " vcpu %u: unexpected "
-                         "exit %u", ctx->plane_id, ctx->vcpu_idx,
-                         kvm_run->exit_reason);
-            ctx->result = -EFAULT;
-            goto done;
-        }
-    }
+static void plane_vcpu_init_cb(CPUState *cs, run_on_cpu_data data)
+{
+    struct plane_vcpu_init_ctx *ctx = data.host_ptr;
 
-done:
-    if (ctx->log_fd >= 0 && ctx->vcpu_idx == 0) {
-        close(ctx->log_fd);
-    }
-    munmap(kvm_run, ctx->vcpu_mmap_size);
-    qemu_mutex_lock(ctx->mutex);
-    if (!boot_signaled) {
-        (*ctx->halted_count)++;
-        qemu_cond_signal(ctx->cond);
-    }
-    qemu_mutex_unlock(ctx->mutex);
-    return NULL;
+    ctx->ret = kvm_init_plane_vcpu(ctx->vcpu_fd, ctx->entry_addr,
+                                   ctx->stack_addr, ctx->zero_page_gpa,
+                                   ctx->page_table_gpa, ctx->gdt_gpa,
+                                   ctx->is_bsp);
 }
 
 static int kvm_handle_hc_vm_planes_activate(X86CPU *cpu, struct kvm_run *run)
@@ -6952,7 +6876,6 @@ static int kvm_handle_hc_vm_planes_activate(X86CPU *cpu, struct kvm_run *run)
     uint64_t plane_count = run->hypercall.args[1];
     uint64_t plane_id;
     KVMState *s = kvm_state;
-    int vcpu_mmap_size;
 
     if (!gpa || !plane_count || !s->vm_planes ||
         plane_count != s->vm_plane_count) {
@@ -6960,26 +6883,13 @@ static int kvm_handle_hc_vm_planes_activate(X86CPU *cpu, struct kvm_run *run)
         return 0;
     }
 
-    vcpu_mmap_size = kvm_ioctl(s, KVM_GET_VCPU_MMAP_SIZE, 0);
-    if (vcpu_mmap_size <= 0) {
-        error_report("vm_planes: KVM_GET_VCPU_MMAP_SIZE failed");
-        run->hypercall.ret = -EINVAL;
-        return 0;
-    }
-
     for (plane_id = 1; plane_id < plane_count; plane_id++) {
         struct kvm_vm_plane_state *ps = &s->vm_planes[plane_id];
         uint64_t stack_addr;
         uint64_t entry_point = 0;
         uint64_t cmdline_gpa, zero_page_gpa;
         uint64_t pt_base, pml4_gpa, pdpt_gpa, pd_base, gdt_gpa_val;
-        struct vm_plane_boot_ctx *ctxs;
-        QemuThread *threads;
-        QemuMutex mutex;
-        QemuCond cond;
-        unsigned int halted_count = 0;
         unsigned int i;
-        int plane_log_fd = -1;
 
         if (kvm_get_plane_fd(s, plane_id) < 0 || !ps->vcpu_count ||
             !ps->host_addr) {
@@ -7086,98 +6996,60 @@ static int kvm_handle_hc_vm_planes_activate(X86CPU *cpu, struct kvm_run *run)
             memcpy(PLANE_HOST(gdt_gpa_val), gdt, sizeof(gdt));
         }
 
-        /* Initialize all plane vCPUs */
+        /* Initialize all plane vCPUs on their owning CPU threads. */
         for (i = 0; i < ps->vcpu_count; i++) {
-            int ret = kvm_init_plane_vcpu(ps->vcpu_fds[i], entry_point,
-                                          stack_addr, zero_page_gpa, pml4_gpa,
-                                          gdt_gpa_val, i == 0);
-            if (ret) {
-                error_report("vm_planes: init plane %" PRIu64 " vcpu %u "
-                             "failed", plane_id, i);
-                run->hypercall.ret = ret;
+            CPUState *target = qemu_get_cpu(ps->vcpu_cpu_index[i]);
+            struct plane_vcpu_init_ctx ictx = {
+                .vcpu_fd = ps->vcpu_fds[i],
+                .entry_addr = entry_point,
+                .stack_addr = stack_addr,
+                .zero_page_gpa = zero_page_gpa,
+                .page_table_gpa = pml4_gpa,
+                .gdt_gpa = gdt_gpa_val,
+                .is_bsp = (i == 0),
+                .ret = -EINVAL,
+            };
+
+            if (!target) {
+                error_report("vm_planes: plane %" PRIu64 " no CPU for vcpu %u",
+                             plane_id, i);
+                run->hypercall.ret = -EINVAL;
                 return 0;
             }
-        }
-#undef PLANE_HOST
-
-        /* Serial log */
-        {
-            char lp[256];
-            snprintf(lp, sizeof(lp), "/tmp/plane%" PRIu64 "-serial.log",
-                     plane_id);
-            plane_log_fd = open(lp, O_CREAT | O_WRONLY | O_TRUNC, 0644);
-        }
 
-        /* Spawn vCPU threads */
-        qemu_mutex_init(&mutex);
-        qemu_cond_init(&cond);
-
-        ctxs = g_new0(struct vm_plane_boot_ctx, ps->vcpu_count);
-        threads = g_new0(QemuThread, ps->vcpu_count);
+            /* See plane_vcpu_init_cb: must run on the sibling's own thread. */
+            bql_lock();
+            run_on_cpu(target, plane_vcpu_init_cb,
+                       RUN_ON_CPU_HOST_PTR(&ictx));
+            bql_unlock();
 
-        for (i = 0; i < ps->vcpu_count; i++) {
-            char name[48];
-
-            ctxs[i].vcpu_fd        = ps->vcpu_fds[i];
-            ctxs[i].vcpu_mmap_size = vcpu_mmap_size;
-            ctxs[i].vcpu_idx       = i;
-            ctxs[i].plane_id       = plane_id;
-            ctxs[i].halted_count   = &halted_count;
-            ctxs[i].mutex          = &mutex;
-            ctxs[i].cond           = &cond;
-            ctxs[i].result         = 0;
-            ctxs[i].log_fd         = (i == 0) ? plane_log_fd : -1;
-            qemu_mutex_init(&ctxs[i].wake_mutex);
-            qemu_cond_init(&ctxs[i].wake_cond);
-            ctxs[i].halted  = false;
-            ctxs[i].kick    = false;
-            ctxs[i].stopped = false;
-
-            snprintf(name, sizeof(name), "plane%" PRIu64 "-vcpu%u",
-                     plane_id, i);
-            qemu_thread_create(&threads[i], name, vm_plane_vcpu_thread,
-                               &ctxs[i], QEMU_THREAD_JOINABLE);
-        }
-
-        /* Wait up to 5s for plane to reach first HLT */
-        {
-            int64_t dl = qemu_clock_get_ns(QEMU_CLOCK_REALTIME) +
-                         5LL * 1000000000LL;
-            qemu_mutex_lock(&mutex);
-            while (halted_count < ps->vcpu_count) {
-                int64_t now = qemu_clock_get_ns(QEMU_CLOCK_REALTIME);
-                if (now >= dl) {
-                    info_report("vm_planes: plane %" PRIu64 " boot timeout "
-                                "(%u/%u halted)", plane_id, halted_count,
-                                ps->vcpu_count);
-                    break;
-                }
-                qemu_cond_timedwait(&cond, &mutex, 1000);
+            if (ictx.ret) {
+                error_report("vm_planes: init plane %" PRIu64 " vcpu %u "
+                             "failed", plane_id, i);
+                run->hypercall.ret = ictx.ret;
+                return 0;
             }
-            qemu_mutex_unlock(&mutex);
         }
+#undef PLANE_HOST
 
-        /* Note: threads keep running for the plane's lifetime; we
-         * intentionally leak ctxs/threads — they outlive this call. */
-
-        /* Seal plane memory via KVM_SET_MEMORY_ATTRIBUTES(NO_WRITE|NO_EXEC) */
-        {
-            struct kvm_memory_attributes ma = {
-                .address    = ps->load_offset,
-                .size       = ps->memory_size,
-                .attributes = KVM_MEMORY_ATTRIBUTE_NO_WRITE |
-                              KVM_MEMORY_ATTRIBUTE_NO_EXEC,
-                .flags      = 0,
-            };
-            int pr = kvm_vm_ioctl(s, KVM_SET_MEMORY_ATTRIBUTES, &ma);
-            if (pr < 0) {
-                warn_report("vm_planes: plane %" PRIu64 " seal failed (%d)",
-                            plane_id, pr);
-            } else {
-                info_report("vm_planes: plane %" PRIu64 " memory sealed",
-                            plane_id);
-            }
-        }
+        /*
+         * Full Option B: do NOT run the secure plane from a dedicated
+         * userspace thread.  The plane's vCPU has been initialized (entry
+         * point, page tables, MP state) but is left STOPPED.  It boots
+         * lazily and in-kernel the first time the normal plane issues a
+         * VBS/VTL call: KVM switches to the secure plane within the normal
+         * plane's KVM_RUN, the secure kernel boots to its dispatch loop and
+         * parks itself via KVM_HC_VBS_VTL_RETURN.  This removes the racy
+         * per-plane thread that concurrently drove the shared vcpu->run
+         * page and caused host lockups.
+         *
+         * NB: the secure plane's own memory is intentionally NOT sealed
+         * here.  Because memory attributes are currently applied to every
+         * plane's EPT alike, sealing the secure plane's region NO_EXEC
+         * would prevent the secure kernel from executing its own dispatch
+         * loop.  Protecting the secure plane's memory from the normal plane
+         * requires plane-aware EPT attributes and is left as a follow-up.
+         */
         ps->host_addr = NULL;
 
         info_report("vm_planes: plane %" PRIu64 " launched — entry 0x%" PRIx64
@@ -7188,107 +7060,13 @@ static int kvm_handle_hc_vm_planes_activate(X86CPU *cpu, struct kvm_run *run)
     return 0;
 }
 
-static int vbs_apply_protection(KVMState *s, uint64_t gpa, uint64_t size,
-                                uint32_t perms)
-{
-    uint64_t attrs = 0;
-    struct kvm_memory_attributes ma;
-
-    if (!(perms & 2)) {
-        attrs |= KVM_MEMORY_ATTRIBUTE_NO_WRITE;
-    }
-    if (!(perms & 4)) {
-        attrs |= KVM_MEMORY_ATTRIBUTE_NO_EXEC;
-    }
-    if (!attrs) {
-        return 0;
-    }
-
-    ma.address    = gpa;
-    ma.size       = size;
-    ma.attributes = attrs;
-    ma.flags      = 0;
-    return kvm_vm_ioctl(s, KVM_SET_MEMORY_ATTRIBUTES, &ma);
-}
-
-static int32_t vbs_handle_protect_memory(KVMState *s, uint64_t ca_gpa)
-{
-    uint64_t gpa, size;
-    uint32_t perms, arg_size;
-
-    cpu_physical_memory_read(ca_gpa + CA_OFF_ARG_SIZE, &arg_size, 4);
-    if (arg_size < 24) {
-        return -22;
-    }
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 0,  &gpa,   8);
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 8,  &size,  8);
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 16, &perms, 4);
-    return vbs_apply_protection(s, gpa, size, perms);
-}
-
-static int32_t vbs_handle_seal_kernel(KVMState *s, uint64_t ca_gpa)
-{
-    uint64_t text_gpa, text_size, rodata_gpa, rodata_size;
-    uint32_t arg_size;
-    int ret;
-
-    cpu_physical_memory_read(ca_gpa + CA_OFF_ARG_SIZE, &arg_size, 4);
-    if (arg_size < 40) {
-        return -22;
-    }
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 0,  &text_gpa,    8);
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 8,  &text_size,   8);
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 16, &rodata_gpa,  8);
-    cpu_physical_memory_read(ca_gpa + CA_OFF_BUFFER + 24, &rodata_size, 8);
-
-    ret = vbs_apply_protection(s, text_gpa, text_size, 1 | 4);
-    if (ret < 0) {
-        return ret;
-    }
-    return vbs_apply_protection(s, rodata_gpa, rodata_size, 1);
-}
-
-static int kvm_handle_hc_vbs_vtl_call(X86CPU *cpu, struct kvm_run *run)
-{
-    uint64_t ca_gpa = run->hypercall.args[0];
-    uint32_t call_id;
-    int32_t status;
-    KVMState *s = kvm_state;
-
-    cpu_physical_memory_read(ca_gpa + CA_OFF_CALL_ID, &call_id, 4);
-
-    switch (call_id) {
-    case VBS_CALL_INIT:
-    case VBS_CALL_SHUTDOWN:
-        status = 0;
-        break;
-    case VBS_CALL_PROTECT_MEMORY:
-        status = vbs_handle_protect_memory(s, ca_gpa);
-        break;
-    case VBS_CALL_SEAL_KERNEL:
-        status = vbs_handle_seal_kernel(s, ca_gpa);
-        break;
-    case VBS_CALL_VALIDATE_MODULE:
-    case VBS_CALL_SET_MODULE_PERMS:
-    case VBS_CALL_UNLOAD_MODULE:
-    case VBS_CALL_ADD_KEY:
-    case VBS_CALL_REVOKE_KEY:
-    case VBS_CALL_SEND_CERTS:
-    case VBS_CALL_KEXEC_VALIDATE:
-    case VBS_CALL_KEXEC_INVALIDATE:
-        status = 0;  /* acknowledge */
-        break;
-    default:
-        warn_report("vbs_vtl_call: unknown call_id 0x%04x", call_id);
-        status = -38;
-        break;
-    }
-
-    cpu_physical_memory_write(ca_gpa + CA_OFF_STATUS, &status, 4);
-    run->hypercall.ret = 0;
-    return 0;
-}
-
+/*
+ * Full Option B: VBS/VTL calls (KVM_HC_VBS_VTL_CALL / _RETURN) are now
+ * serviced entirely in-kernel by switching to the secure plane, which runs
+ * a real in-guest dispatcher.  QEMU no longer emulates the secure world, so
+ * those hypercalls never exit to userspace here.  Only the plane
+ * configuration/activation hypercalls are handled by QEMU.
+ */
 static int kvm_handle_hypercall(X86CPU *cpu, struct kvm_run *run)
 {
     if (run->hypercall.nr == KVM_HC_MAP_GPA_RANGE)
@@ -7297,8 +7075,6 @@ static int kvm_handle_hypercall(X86CPU *cpu, struct kvm_run *run)
         return kvm_handle_hc_vm_planes_config(cpu, run);
     if (run->hypercall.nr == KVM_HC_VM_PLANES_ACTIVATE)
         return kvm_handle_hc_vm_planes_activate(cpu, run);
-    if (run->hypercall.nr == KVM_HC_VBS_VTL_CALL)
-        return kvm_handle_hc_vbs_vtl_call(cpu, run);
 
     return -EINVAL;
 }
-- 
2.55.0


  parent reply	other threads:[~2026-08-05 11:04 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-05 11:04 [RFC PATCH v1 0/5] VBS/VSM-on-KVM: QEMU support for the secure VM plane Sriram Nambakam
2026-08-05 11:04 ` [RFC PATCH v1 1/5] kvm: add userspace handlers for VM planes and VBS VTL calls Sriram Nambakam
2026-08-05 11:04 ` [RFC PATCH v1 2/5] vm_planes: Add VBS VTL call handling and plane memory sealing Sriram Nambakam
2026-08-05 11:04 ` [RFC PATCH v1 3/5] linux-headers: sync kvm_para.h VBS VTL hypercalls Sriram Nambakam
2026-08-05 11:04 ` Sriram Nambakam [this message]
2026-08-05 11:04 ` [RFC PATCH v1 5/5] target/i386/kvm: read plane config via address_space API Sriram Nambakam

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260805110432.25167-5-snambakam@linux.microsoft.com \
    --to=snambakam@linux.microsoft.com \
    --cc=kvm@vger.kernel.org \
    --cc=qemu-devel@nongnu.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox