From: <mhonap@nvidia.com>
To: <alex@shazbot.org>, <jgg@ziepe.ca>, <ankita@nvidia.com>,
<jic23@kernel.org>, <dave.jiang@intel.com>,
<alejandro.lucero-palau@amd.com>, <smadhavan@nvidia.com>,
<corbet@lwn.net>, <skhan@linuxfoundation.org>,
<dave@stgolabs.net>, <alison.schofield@intel.com>,
<vishal.l.verma@intel.com>, <iweiny@kernel.org>,
<ming.li@zohomail.com>, <yishaih@nvidia.com>,
<skolothumtho@nvidia.com>, <kevin.tian@intel.com>,
<bhelgaas@google.com>, <dmatlack@google.com>, <kees@kernel.org>,
<gustavoars@kernel.org>
Cc: <cjia@nvidia.com>, <kjaju@nvidia.com>, <vsethi@nvidia.com>,
<zhiw@nvidia.com>, <mhonap@nvidia.com>,
<linux-doc@vger.kernel.org>, <linux-kernel@vger.kernel.org>,
<kvm@vger.kernel.org>, <linux-cxl@vger.kernel.org>,
<linux-pci@vger.kernel.org>, <linux-kselftest@vger.kernel.org>,
<linux-hardening@vger.kernel.org>
Subject: [PATCH v4 12/27] vfio/pci: Let a provider exclude a BAR sub-range from mmap
Date: Thu, 13 Aug 2026 15:06:16 +0530 [thread overview]
Message-ID: <20260813093631.2288172-13-mhonap@nvidia.com> (raw)
In-Reply-To: <20260813093631.2288172-1-mhonap@nvidia.com>
From: Manish Honap <mhonap@nvidia.com>
Some devices expose registers in a BAR that must be reached only through
a trap, not a direct guest mapping. A CXL Type-2 device's HDM decoder
block is one: mapping it would let userspace reprogram the physical
decoder that governs host memory decode. Give a provider a way to mark a
BAR sub-range off-limits to mmap; it is advertised as a sparse-mmap
region and refused in the mmap path, while the provider's own region
still serves it.
Signed-off-by: Manish Honap <mhonap@nvidia.com>
---
drivers/vfio/pci/vfio_pci_core.c | 72 ++++++++++++++++++++++++++++++
drivers/vfio/pci/vfio_pci_dmabuf.c | 13 ++++++
drivers/vfio/pci/vfio_pci_priv.h | 13 ++++++
include/linux/vfio_pci_core.h | 6 +++
4 files changed, 104 insertions(+)
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 0f9b5dfeea66..49dfbdaf3f05 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -1009,6 +1009,67 @@ static int msix_mmappable_cap(struct vfio_pci_core_device *vdev,
return vfio_info_add_capability(caps, &header, sizeof(header));
}
+/*
+ * A provider can keep a BAR sub-range off mmap (for example a CXL device's
+ * trapped HDM decoder block). Callers hold the resource so /dev/mem is already
+ * blocked; this only governs the vfio mmap path.
+ */
+void vfio_pci_core_set_mmap_exclude(struct vfio_pci_core_device *vdev, int bar,
+ u64 start, u64 len)
+{
+ vdev->mmap_exclude_bar = bar;
+ vdev->mmap_exclude_start = start;
+ vdev->mmap_exclude_len = len;
+}
+EXPORT_SYMBOL_GPL(vfio_pci_core_set_mmap_exclude);
+
+/* Advertise the BAR as mmappable minus the excluded sub-range. */
+static int vfio_pci_mmap_exclude_cap(struct vfio_pci_core_device *vdev,
+ int index, struct vfio_info_cap *caps)
+{
+ u64 bar_len = pci_resource_len(vdev->pdev, index);
+ u64 excl_start = ALIGN_DOWN(vdev->mmap_exclude_start, PAGE_SIZE);
+ u64 excl_end = ALIGN(vdev->mmap_exclude_start + vdev->mmap_exclude_len,
+ PAGE_SIZE);
+ struct vfio_region_info_cap_sparse_mmap *sparse;
+ int nr_areas = 0, i = 0, ret;
+ size_t size;
+
+ /*
+ * mmap is page granular, so the mmappable areas must stop at the page
+ * boundaries enclosing the excluded sub-range. The byte-granular
+ * exclusion still governs the fault and read/write paths; only the
+ * advertised mmap areas round out to whole pages.
+ */
+ if (excl_start > 0)
+ nr_areas++;
+ if (excl_end < bar_len)
+ nr_areas++;
+
+ size = struct_size(sparse, areas, nr_areas);
+ sparse = kzalloc(size, GFP_KERNEL);
+ if (!sparse)
+ return -ENOMEM;
+
+ sparse->header.id = VFIO_REGION_INFO_CAP_SPARSE_MMAP;
+ sparse->header.version = 1;
+ sparse->nr_areas = nr_areas;
+
+ if (excl_start > 0) {
+ sparse->areas[i].offset = 0;
+ sparse->areas[i].size = excl_start;
+ i++;
+ }
+ if (excl_end < bar_len) {
+ sparse->areas[i].offset = excl_end;
+ sparse->areas[i].size = bar_len - excl_end;
+ }
+
+ ret = vfio_info_add_capability(caps, &sparse->header, size);
+ kfree(sparse);
+ return ret;
+}
+
int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
unsigned int type, unsigned int subtype,
const struct vfio_pci_regops *ops,
@@ -1157,6 +1218,13 @@ int vfio_pci_ioctl_get_region_info(struct vfio_device *core_vdev,
if (ret)
return ret;
}
+ if (vdev->mmap_exclude_len &&
+ info->index == vdev->mmap_exclude_bar) {
+ ret = vfio_pci_mmap_exclude_cap(vdev, info->index,
+ caps);
+ if (ret)
+ return ret;
+ }
}
break;
@@ -1851,6 +1919,10 @@ int vfio_pci_core_mmap(struct vfio_device *core_vdev, struct vm_area_struct *vma
if (req_start + req_len > phys_len)
return -EINVAL;
+ /* An excluded sub-range is reachable only through its trap, not mmap. */
+ if (vfio_pci_bar_is_excluded(vdev, index, req_start, req_len))
+ return -EINVAL;
+
/*
* Ensure the BAR resource region is reserved for use.
*/
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c
index c16f460c01d6..51983105d38b 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -177,11 +177,24 @@ int vfio_pci_core_get_dmabuf_phys(struct vfio_pci_core_device *vdev,
size_t nr_ranges)
{
struct pci_dev *pdev = vdev->pdev;
+ unsigned int i;
*provider = pcim_p2pdma_provider(pdev, region_index);
if (!*provider)
return -EINVAL;
+ /*
+ * A provider (e.g. vfio-cxl) can exclude a BAR sub-range that must be
+ * reached only through its trap. The mmap and read/write paths already
+ * refuse it; reject a DMA-BUF export overlapping it too, so a device fd
+ * holder cannot map the excluded registers to a peer and bypass the trap.
+ */
+ for (i = 0; i < nr_ranges; i++)
+ if (vfio_pci_bar_is_excluded(vdev, region_index,
+ dma_ranges[i].offset,
+ dma_ranges[i].length))
+ return -EINVAL;
+
return vfio_pci_core_fill_phys_vec(
phys_vec, dma_ranges, nr_ranges,
pci_resource_start(pdev, region_index),
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index fca9d0dfac90..902d17815ab6 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -44,6 +44,19 @@ ssize_t vfio_pci_config_rw_single(struct vfio_pci_core_device *vdev,
ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
size_t count, loff_t *ppos, bool iswrite);
+/*
+ * A provider (e.g. vfio-cxl) can carve a sub-range out of a BAR that must be
+ * reached only through its trap, never the direct BAR. Returns true when
+ * [start, start + len) on this BAR overlaps that excluded range.
+ */
+static inline bool vfio_pci_bar_is_excluded(struct vfio_pci_core_device *vdev,
+ int bar, u64 start, u64 len)
+{
+ return vdev->mmap_exclude_len && bar == vdev->mmap_exclude_bar &&
+ start < vdev->mmap_exclude_start + vdev->mmap_exclude_len &&
+ start + len > vdev->mmap_exclude_start;
+}
+
#ifdef CONFIG_VFIO_PCI_VGA
ssize_t vfio_pci_vga_rw(struct vfio_pci_core_device *vdev, char __user *buf,
size_t count, loff_t *ppos, bool iswrite);
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 117cd67995d8..43755b91880f 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -162,6 +162,10 @@ struct vfio_pci_core_device {
struct notifier_block nb;
struct rw_semaphore memory_lock;
struct list_head dmabufs;
+ /* BAR sub-range a provider keeps off mmap, reached only through a trap */
+ int mmap_exclude_bar;
+ u64 mmap_exclude_start;
+ u64 mmap_exclude_len;
};
enum vfio_pci_io_width {
@@ -176,6 +180,8 @@ int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
unsigned int type, unsigned int subtype,
const struct vfio_pci_regops *ops,
size_t size, u32 flags, void *data);
+void vfio_pci_core_set_mmap_exclude(struct vfio_pci_core_device *vdev, int bar,
+ u64 start, u64 len);
void vfio_pci_core_close_device(struct vfio_device *core_vdev);
int vfio_pci_core_init_dev(struct vfio_device *core_vdev);
void vfio_pci_core_release_dev(struct vfio_device *core_vdev);
--
2.25.1
next prev parent reply other threads:[~2026-08-13 9:39 UTC|newest]
Thread overview: 28+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-13 9:36 [PATCH v4 00/27] vfio/pci: Add CXL Type-2 device passthrough support mhonap
2026-08-13 9:36 ` [PATCH v4 01/27] cxl: Fix resource.c include path and export cxl_restore_hdm_after_pci_reset mhonap
2026-08-13 9:36 ` [PATCH v4 02/27] cxl/regs: Skip sub-block region request for BAR-owning drivers mhonap
2026-08-13 9:36 ` [PATCH v4 03/27] cxl: Move component register defines to uapi/cxl/cxl_regs.h mhonap
2026-08-13 9:36 ` [PATCH v4 04/27] cxl: Establish media readiness in cxl_mem_probe() mhonap
2026-08-13 9:36 ` [PATCH v4 05/27] cxl: Add a function-scoped reset entry for vfio-pci mhonap
2026-08-13 9:36 ` [PATCH v4 06/27] vfio/pci: Add CXL ops registration interface mhonap
2026-08-13 9:36 ` [PATCH v4 07/27] vfio/pci: Detect CXL devices and load vfio-cxl on demand mhonap
2026-08-13 9:36 ` [PATCH v4 08/27] vfio/cxl: Add the vfio-cxl module skeleton mhonap
2026-08-13 9:36 ` [PATCH v4 09/27] vfio/cxl: Create the CXL memory device at bind mhonap
2026-08-13 9:36 ` [PATCH v4 10/27] vfio/cxl: Reject unsupported decoder topologies " mhonap
2026-08-13 9:36 ` [PATCH v4 11/27] vfio/cxl: Own the whole component register BAR mhonap
2026-08-13 9:36 ` mhonap [this message]
2026-08-13 9:36 ` [PATCH v4 13/27] vfio/pci: Refuse read/write to an excluded BAR sub-range mhonap
2026-08-13 9:36 ` [PATCH v4 14/27] vfio: Add CXL region type for the HDM region mhonap
2026-08-13 9:36 ` [PATCH v4 15/27] vfio/pci: Call CXL open and close hooks around device use mhonap
2026-08-13 9:36 ` [PATCH v4 16/27] vfio/cxl: Shadow the CXL DVSEC body at open mhonap
2026-08-13 9:36 ` [PATCH v4 17/27] vfio/cxl: Virtualize the CXL DVSEC mhonap
2026-08-13 9:36 ` [PATCH v4 18/27] vfio/cxl: Expose the HDM memory and trap the decoder registers mhonap
2026-08-13 9:36 ` [PATCH v4 19/27] vfio/cxl: Keep the HDM decoder block off the direct BAR mapping mhonap
2026-08-13 9:36 ` [PATCH v4 20/27] vfio/cxl: Emulate the HDM decoder commit handshake mhonap
2026-08-13 9:36 ` [PATCH v4 21/27] vfio/cxl: Describe the CXL device and decoder geometry to userspace mhonap
2026-08-13 9:36 ` [PATCH v4 22/27] vfio/cxl: Revoke the HDM mapping on reset and power transitions mhonap
2026-08-13 9:36 ` [PATCH v4 23/27] vfio/cxl: Refresh the decoder snapshot after a device reset mhonap
2026-08-13 9:36 ` [PATCH v4 24/27] vfio/cxl: Service a guest-triggered CXL reset mhonap
2026-08-13 9:36 ` [PATCH v4 25/27] vfio/pci: Provide an opt-out for the CXL Type-2 extensions mhonap
2026-08-13 9:36 ` [PATCH v4 26/27] Documentation: vfio-pci: Document CXL Type-2 device passthrough mhonap
2026-08-13 9:36 ` [PATCH v4 27/27] selftests/vfio: Add CXL Type-2 passthrough corner-case tests mhonap
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260813093631.2288172-13-mhonap@nvidia.com \
--to=mhonap@nvidia.com \
--cc=alejandro.lucero-palau@amd.com \
--cc=alex@shazbot.org \
--cc=alison.schofield@intel.com \
--cc=ankita@nvidia.com \
--cc=bhelgaas@google.com \
--cc=cjia@nvidia.com \
--cc=corbet@lwn.net \
--cc=dave.jiang@intel.com \
--cc=dave@stgolabs.net \
--cc=dmatlack@google.com \
--cc=gustavoars@kernel.org \
--cc=iweiny@kernel.org \
--cc=jgg@ziepe.ca \
--cc=jic23@kernel.org \
--cc=kees@kernel.org \
--cc=kevin.tian@intel.com \
--cc=kjaju@nvidia.com \
--cc=kvm@vger.kernel.org \
--cc=linux-cxl@vger.kernel.org \
--cc=linux-doc@vger.kernel.org \
--cc=linux-hardening@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=ming.li@zohomail.com \
--cc=skhan@linuxfoundation.org \
--cc=skolothumtho@nvidia.com \
--cc=smadhavan@nvidia.com \
--cc=vishal.l.verma@intel.com \
--cc=vsethi@nvidia.com \
--cc=yishaih@nvidia.com \
--cc=zhiw@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox