All of lore.kernel.org
 help / color / mirror / Atom feed
From: Narayana Murty N <nnmlinux@linux.ibm.com>
To: qemu-devel@nongnu.org, qemu-ppc@nongnu.org, sbhat@linux.ibm.com,
	mahesh@linux.ibm.com, sourabhjain@linux.ibm.com
Cc: npiggin@gmail.com, harshpb@linux.ibm.com, amachhiw@linux.ibm.com,
	adityag@linux.ibm.com, hbathini@linux.ibm.com,
	shivangu@linux.ibm.com, anushree.mathur@linux.vnet.ibm.com
Subject: [PATCH v1 2/2] ppc/spapr: Temporarily disable VFIO BAR mmap during EEH PE reset
Date: Wed,  2 Sep 2026 17:17:58 +0530	[thread overview]
Message-ID: <20260902114758.85160-3-nnmlinux@linux.ibm.com> (raw)
In-Reply-To: <20260902114758.85160-1-nnmlinux@linux.ibm.com>

An EEH PE hot or fundamental reset involves a synchronous kernel ioctl
(VFIO_EEH_PE_RESET_HOT / VFIO_EEH_PE_RESET_FUNDAMENTAL) that asserts
then de-asserts a PCI reset signal to the endpoint.  During this window
the device's BARs are inaccessible, but QEMU may still have those BARs
memory-mapped for direct guest access.  A guest MMIO fault that arrives
while the hardware is in reset can cause an unexpected host kernel page
fault or an indeterminate read value.

Fix this by disabling the BAR mmap windows for every VFIO PCI device
under the PHB before issuing the PE reset ioctl, and re-enabling them
after VFIO_EEH_PE_CONFIGURE succeeds.

A new spapr_phb_vfio_eeh_post_configure() bus walker calls
vfio_region_mmaps_set_enabled(..., true) on the configure success
path inside spapr_phb_vfio_eeh_configure().  On failure the mmaps
remain disabled; the next EEH reset attempt will call pre_reset
again.

Signed-off-by: Narayana Murty N <nnmlinux@linux.ibm.com>
---
 hw/ppc/spapr_pci_vfio.c | 68 +++++++++++++++++++++++++++++++++++++++++
 1 file changed, 68 insertions(+)

diff --git a/hw/ppc/spapr_pci_vfio.c b/hw/ppc/spapr_pci_vfio.c
index c233822d14..1cf058cbde 100644
--- a/hw/ppc/spapr_pci_vfio.c
+++ b/hw/ppc/spapr_pci_vfio.c
@@ -252,17 +252,28 @@ int spapr_phb_vfio_eeh_get_state(SpaprPhbState *sphb, int *state)
  * pci_host_config_write_common() so that the VFIO config-write handler calls
  * vfio_msix_disable(), cleanly releasing vectors and KVM irqfd routes while
  * leaving the shadow intact.
+ *
+ * After disabling interrupts, disable BAR mmap windows so that the host
+ * kernel PE reset ioctl does not race with QEMU direct-mapped guest accesses.
+ * The timer that would ordinarily re-enable mmaps after an INTx quiet period
+ * is cancelled here; mmaps are restored after VFIO_EEH_PE_CONFIGURE succeeds
+ * in spapr_phb_vfio_eeh_configure().
  */
 static void spapr_phb_vfio_eeh_prepare_dev(PCIBus *bus,
                                            PCIDevice *pdev,
                                            void *opaque)
 {
+    VFIOPCIDevice *vdev;
     uint16_t flags;
+    int i;
 
     if (!object_dynamic_cast(OBJECT(pdev), TYPE_VFIO_PCI_DEVICE)) {
         return;
     }
 
+    vdev = VFIO_PCI_DEVICE(pdev);
+
+    /* Step 1: disable MSI-X without wiping the shadow table (see above). */
     if (msix_enabled(pdev)) {
         flags = pci_get_word(pdev->config + pdev->msix_cap + PCI_MSIX_FLAGS);
         flags &= ~PCI_MSIX_FLAGS_ENABLE;
@@ -270,6 +281,19 @@ static void spapr_phb_vfio_eeh_prepare_dev(PCIBus *bus,
                                      pdev->msix_cap + PCI_MSIX_FLAGS,
                                      pci_config_size(pdev), flags, 2);
     }
+
+    /*
+     * Step 2: cancel any pending INTx mmap re-enable timer.  The timer is
+     * only allocated when PCI_INTERRUPT_PIN is non-zero, so guard the call.
+     */
+    if (vdev->intx.mmap_timer) {
+        timer_del(vdev->intx.mmap_timer);
+    }
+
+    /* Step 3: disable BAR mmaps last, after interrupt teardown. */
+    for (i = 0; i < PCI_ROM_SLOT; i++) {
+        vfio_region_mmaps_set_enabled(&vdev->bars[i].region, false);
+    }
 }
 
 static void spapr_phb_vfio_eeh_prepare_bus(PCIBus *bus, void *opaque)
@@ -286,6 +310,43 @@ static void spapr_phb_vfio_eeh_pre_reset(SpaprPhbState *sphb)
     pci_for_each_bus(phb->bus, spapr_phb_vfio_eeh_prepare_bus, NULL);
 }
 
+/*
+ * Re-enable BAR mmap windows for a single VFIO PCI device after a successful
+ * EEH PE configure.  Called only on the configure success path; on failure the
+ * mmaps remain disabled until the next hot/fundamental reset attempt.
+ */
+static void spapr_phb_vfio_eeh_post_configure_dev(PCIBus *bus,
+                                                   PCIDevice *pdev,
+                                                   void *opaque)
+{
+    VFIOPCIDevice *vdev;
+    int i;
+
+    if (!object_dynamic_cast(OBJECT(pdev), TYPE_VFIO_PCI_DEVICE)) {
+        return;
+    }
+
+    vdev = VFIO_PCI_DEVICE(pdev);
+
+    for (i = 0; i < PCI_ROM_SLOT; i++) {
+        vfio_region_mmaps_set_enabled(&vdev->bars[i].region, true);
+    }
+}
+
+static void spapr_phb_vfio_eeh_post_configure_bus(PCIBus *bus, void *opaque)
+{
+    pci_for_each_device_under_bus(bus,
+                                  spapr_phb_vfio_eeh_post_configure_dev,
+                                  NULL);
+}
+
+static void spapr_phb_vfio_eeh_post_configure(SpaprPhbState *sphb)
+{
+    PCIHostState *phb = PCI_HOST_BRIDGE(sphb);
+
+    pci_for_each_bus(phb->bus, spapr_phb_vfio_eeh_post_configure_bus, NULL);
+}
+
 int spapr_phb_vfio_eeh_reset(SpaprPhbState *sphb, int option)
 {
     uint32_t op;
@@ -293,6 +354,11 @@ int spapr_phb_vfio_eeh_reset(SpaprPhbState *sphb, int option)
 
     switch (option) {
     case RTAS_SLOT_RESET_DEACTIVATE:
+        /*
+         * Deactivate does not perform a full PE reset; BAR mmaps were already
+         * disabled by the preceding HOT or FUNDAMENTAL reset call and must not
+         * be re-enabled here.
+         */
         op = VFIO_EEH_PE_RESET_DEACTIVATE;
         break;
     case RTAS_SLOT_RESET_HOT:
@@ -324,6 +390,8 @@ int spapr_phb_vfio_eeh_configure(SpaprPhbState *sphb)
         return RTAS_OUT_PARAM_ERROR;
     }
 
+    spapr_phb_vfio_eeh_post_configure(sphb);
+
     return RTAS_OUT_SUCCESS;
 }
 
-- 
2.51.1



      parent reply	other threads:[~2026-09-02 11:49 UTC|newest]

Thread overview: 3+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-02 11:47 [PATCH 0/2] ppc/spapr: Fix MSI-X shadow and BAR mmap handling during EEH PE reset Narayana Murty N
2026-09-02 11:47 ` [PATCH v1 1/2] ppc/spapr: Preserve MSI-X shadow across " Narayana Murty N
2026-09-02 11:47 ` Narayana Murty N [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260902114758.85160-3-nnmlinux@linux.ibm.com \
    --to=nnmlinux@linux.ibm.com \
    --cc=adityag@linux.ibm.com \
    --cc=amachhiw@linux.ibm.com \
    --cc=anushree.mathur@linux.vnet.ibm.com \
    --cc=harshpb@linux.ibm.com \
    --cc=hbathini@linux.ibm.com \
    --cc=mahesh@linux.ibm.com \
    --cc=npiggin@gmail.com \
    --cc=qemu-devel@nongnu.org \
    --cc=qemu-ppc@nongnu.org \
    --cc=sbhat@linux.ibm.com \
    --cc=shivangu@linux.ibm.com \
    --cc=sourabhjain@linux.ibm.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.